From 8b3ed0a400ad0bbd5d567e313b6cf5358178b723 Mon Sep 17 00:00:00 2001 From: Daniel Han Date: Thu, 27 Aug 2026 02:35:16 +0000 Subject: [PATCH 1/9] qwen4exp: keep the indexer cache in step across server slots The QSA indexer keeps a side cache addressed by the cells of the attention cache, so cell j has to hold the same token in both: the top-k indices it produces are applied to the attention KQ mask. init_batch already hands the indexer the attention cache's slot layout rather than letting it look for its own, but the restore path did not. state_read called llama_kv_cache::state_read on the two caches in turn and each ran its own find_slot over its own occupancy. That agrees only for as long as nothing has already pushed the two caches apart, which is the property a restore is supposed to re-establish rather than one it can lean on. The failure path was the worse half, and it is reachable from the public API with nothing more than a short buffer. Truncating a good blob at 35 offsets and feeding it to llama_state_seq_set_data left the two caches disagreeing at 5 of them, and every one of 23 truncations of a whole-context blob did. Four of those five land inside the attention section, so the attention cache drops the sequence and the indexer keeps it; only the cut that lands in the indexer section gives the opposite direction. llama_kv_cache::state_read cleans up its own cache and rethrows, so whichever way it falls, nothing is left to bring the two back together. The server papers over this by clearing the slot when a prompt cache load fails; a caller of llama_state_seq_set_data that does not is left with an indexer addressing cells that no longer mean what it thinks. llama_kv_cache::state_read_sinfo reports the cells a restore landed in, or takes a copy of them, and state_read_meta uses a supplied layout in place of find_slot once it has checked that those cells are free here too. The indexer now adopts the attention cache's restored layout by construction instead of reproducing it by coincidence, and a layout that does not fit fails the read rather than being applied over cells that already drifted. The hybrid restore is wrapped so that any failure drops the sequence, or for a whole-context restore the context, from all three caches at once, which is a state they do agree on. --- src/llama-kv-cache.cpp | 66 ++++++++++++++++++++++++++++++--- src/llama-kv-cache.h | 18 ++++++++- src/llama-memory-hybrid-idx.cpp | 59 ++++++++++++++++++++++++----- src/llama-memory-hybrid-idx.h | 5 +++ 4 files changed, 132 insertions(+), 16 deletions(-) diff --git a/src/llama-kv-cache.cpp b/src/llama-kv-cache.cpp index f815cd7665..3330b66262 100644 --- a/src/llama-kv-cache.cpp +++ b/src/llama-kv-cache.cpp @@ -2038,6 +2038,15 @@ void llama_kv_cache::state_write(llama_io_write_i & io, llama_seq_id seq_id, lla } void llama_kv_cache::state_read(llama_io_read_i & io, llama_seq_id seq_id, llama_state_seq_flags flags) { + state_read_sinfo(io, seq_id, flags, nullptr, nullptr); +} + +void llama_kv_cache::state_read_sinfo( + llama_io_read_i & io, + llama_seq_id seq_id, + llama_state_seq_flags flags, + slot_info_vec_t * sinfos_out, +const slot_info_vec_t * sinfos_in) { // TODO: refactor [TAG_KV_CACHE_SHARE_CELLS] if (other) { return; @@ -2048,6 +2057,14 @@ void llama_kv_cache::state_read(llama_io_read_i & io, llama_seq_id seq_id, llama // TODO: fix incosistent handling of `seq_id < 0` and `seq_id == -1` in the codebase [TAG_LLAMA_SEQ_ID_NEG] GGML_ASSERT(seq_id == -1 || (seq_id >= 0 && (size_t) seq_id < seq_to_stream.size())); + if (sinfos_out) { + sinfos_out->assign(n_stream, slot_info{}); + } + + if (sinfos_in && sinfos_in->size() != n_stream) { + throw std::runtime_error("failed to restore kv cache: mirrored slot layout has the wrong stream count"); + } + uint32_t n_stream_cur; io.read(&n_stream_cur, sizeof(n_stream_cur)); if (n_stream_cur != n_stream) { @@ -2059,6 +2076,10 @@ void llama_kv_cache::state_read(llama_io_read_i & io, llama_seq_id seq_id, llama io.read(&cell_count, sizeof(cell_count)); if (cell_count == 0) { + // a mirrored cache must be empty here as well, or the two no longer agree cell for cell + if (sinfos_in && !(*sinfos_in)[s].empty()) { + throw std::runtime_error("failed to restore kv cache: mirrored cache holds cells this one does not"); + } continue; } @@ -2067,7 +2088,7 @@ void llama_kv_cache::state_read(llama_io_read_i & io, llama_seq_id seq_id, llama slot_info sinfo; bool res = true; - res = res && state_read_meta(io, strm, cell_count, sinfo, seq_id); + res = res && state_read_meta(io, strm, cell_count, sinfo, seq_id, sinfos_in ? &(*sinfos_in)[s] : nullptr); try { res = res && state_read_data(io, strm, cell_count, sinfo); @@ -2083,6 +2104,10 @@ void llama_kv_cache::state_read(llama_io_read_i & io, llama_seq_id seq_id, llama } throw std::runtime_error("failed to restore kv cache"); } + + if (sinfos_out) { + (*sinfos_out)[s] = sinfo; + } } } @@ -2218,7 +2243,7 @@ void llama_kv_cache::state_write_data(llama_io_write_i & io, const cell_ranges_t } } -bool llama_kv_cache::state_read_meta(llama_io_read_i & io, uint32_t strm, uint32_t cell_count, slot_info & sinfo, llama_seq_id dest_seq_id) { +bool llama_kv_cache::state_read_meta(llama_io_read_i & io, uint32_t strm, uint32_t cell_count, slot_info & sinfo, llama_seq_id dest_seq_id, const slot_info * sinfo_in) { auto & cells = v_cells[strm]; auto & head = v_heads[strm]; @@ -2263,10 +2288,39 @@ bool llama_kv_cache::state_read_meta(llama_io_read_i & io, uint32_t strm, uint32 ubatch.seq_id[i] = &dest_seq_id; } - sinfo = find_slot(ubatch, false); - if (sinfo.empty()) { - LLAMA_LOG_ERROR("%s: failed to find %d available cells in kv cache\n", __func__, cell_count); - return false; + if (sinfo_in) { + // this cache mirrors another one, so it takes that cache's restored layout rather + // than searching for cells of its own + if (sinfo_in->empty() || sinfo_in->n_stream() != 1 || sinfo_in->idxs[0].size() != cell_count) { + LLAMA_LOG_ERROR("%s: mirrored slot layout holds %d cells, this cache restores %d\n", __func__, + sinfo_in->empty() ? 0 : (int) sinfo_in->idxs[0].size(), cell_count); + return false; + } + + sinfo = *sinfo_in; + + // the layout is addressed by cell index, so it only means the same thing in both + // caches while their streams line up + sinfo.s0 = strm; + sinfo.s1 = strm; + sinfo.strm[0] = strm; + + // seq_rm above freed exactly the cells this sequence held. anything else in the way + // is a cache that had already drifted, which this restore must not paper over + for (uint32_t i = 0; i < cell_count; ++i) { + const uint32_t idx = sinfo.idxs[0][i]; + + if (idx >= cells.size() || !cells.is_empty(idx)) { + LLAMA_LOG_ERROR("%s: cell %u of the mirrored slot layout is not free\n", __func__, idx); + return false; + } + } + } else { + sinfo = find_slot(ubatch, false); + if (sinfo.empty()) { + LLAMA_LOG_ERROR("%s: failed to find %d available cells in kv cache\n", __func__, cell_count); + return false; + } } // TODO: we cannot yet restore llama_kv_cell_ext as the apply_ubatch() does not support it yet diff --git a/src/llama-kv-cache.h b/src/llama-kv-cache.h index f00ec2137b..b211ca36f9 100644 --- a/src/llama-kv-cache.h +++ b/src/llama-kv-cache.h @@ -168,6 +168,21 @@ public: const llama_kv_cells & get_cells(llama_seq_id seq_id) const; + // state_read, plus the cells the restored tokens were placed in. + // a cache that mirrors another one cell for cell (the qwen4exp indexer) cannot search for + // its own cells here: a second independent search only happens to agree with the first. + // sinfos_out: if set, resized to n_stream and filled with the layout used; a stream that + // carried no cells leaves an empty entry + // sinfos_in : if set, the layout to use instead of searching for one. it must have one + // entry per stream and the entry must match the cell count in the blob, + // otherwise the read fails as it would on any other corrupt input + void state_read_sinfo( + llama_io_read_i & io, + llama_seq_id seq_id, + llama_state_seq_flags flags, + slot_info_vec_t * sinfos_out, + const slot_info_vec_t * sinfos_in); + // // graph_build API // @@ -320,7 +335,8 @@ private: void state_write_meta(llama_io_write_i & io, const cell_ranges_t & cr, llama_seq_id seq_id = -1) const; void state_write_data(llama_io_write_i & io, const cell_ranges_t & cr) const; - bool state_read_meta(llama_io_read_i & io, uint32_t strm, uint32_t cell_count, slot_info & sinfo, llama_seq_id dest_seq_id = -1); + // sinfo_in, when set, replaces the find_slot call: the cells are given by the caller + bool state_read_meta(llama_io_read_i & io, uint32_t strm, uint32_t cell_count, slot_info & sinfo, llama_seq_id dest_seq_id = -1, const slot_info * sinfo_in = nullptr); bool state_read_data(llama_io_read_i & io, uint32_t strm, uint32_t cell_count, const slot_info & sinfo); }; diff --git a/src/llama-memory-hybrid-idx.cpp b/src/llama-memory-hybrid-idx.cpp index e7f6e565dc..38e3e61c57 100644 --- a/src/llama-memory-hybrid-idx.cpp +++ b/src/llama-memory-hybrid-idx.cpp @@ -476,19 +476,60 @@ void llama_memory_hybrid_idx::state_write(llama_io_write_i & io, llama_seq_id se } void llama_memory_hybrid_idx::state_read(llama_io_read_i & io, llama_seq_id seq_id, llama_state_seq_flags flags) { - llama_memory_hybrid::state_read(io, seq_id, flags); + // note: this repeats llama_memory_hybrid::state_read because the indexer cache has to be + // handed the cells the attention cache restored into, and because a restore that + // fails halfway has to leave all three caches in the same state - // [TAG_HYBRID_IDX_STATE] must mirror the write order above. - // The indexer finds its own cells, which is safe because the two caches stay in lockstep: - // both state_read_meta calls run find_slot over the same occupancy and land on the same cells. - if ((flags & LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY) == 0) { - if (mem_idx) { - mem_idx->state_read(io, seq_id, flags); + // [TAG_HYBRID_IDX_SINFO] + // The indexer cache is addressed by the cells of the attention cache, so its restore adopts + // that layout instead of searching for cells of its own. Two independent find_slot calls + // agree only while nothing makes the two caches see different occupancy, and a restore is + // exactly the operation that can no longer promise that. + llama_kv_cache::slot_info_vec_t sinfos_attn; + + try { + if ((flags & LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY) == 0) { + get_mem_attn()->state_read_sinfo(io, seq_id, flags, mem_idx ? &sinfos_attn : nullptr, nullptr); } + + get_mem_recr()->state_read(io, seq_id, flags); + + // [TAG_HYBRID_IDX_STATE] must mirror the write order in state_write + if ((flags & LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY) == 0) { + if (mem_idx) { + mem_idx->state_read_sinfo(io, seq_id, flags, nullptr, &sinfos_attn); + } + } + + // [TAG_PLE_HISTORY] must mirror the write order above + ple_hist_state_read(io, seq_id); + } catch (...) { + // a half-restored context is the one state the indexer cache cannot be brought back from + // by itself: the attention cache holds the restored cells and the indexer the old ones. + // drop what was being restored from all of them, which is a state they do agree on. + state_drop(seq_id); + + throw; + } +} + +void llama_memory_hybrid_idx::state_drop(llama_seq_id seq_id) { + // dropped directly rather than through seq_rm, which the recurrent cache is allowed to + // refuse and which would then clear the other two caches and not it + if (seq_id < 0) { + clear(true); + + return; } - // [TAG_PLE_HISTORY] must mirror the write order above - ple_hist_state_read(io, seq_id); + get_mem_attn()->seq_rm(seq_id, -1, -1); + get_mem_recr()->seq_rm(seq_id, -1, -1); + + if (mem_idx) { + mem_idx->seq_rm(seq_id, -1, -1); + } + + ple_hist.erase(seq_id); } llama_kv_cache * llama_memory_hybrid_idx::get_mem_idx() const { diff --git a/src/llama-memory-hybrid-idx.h b/src/llama-memory-hybrid-idx.h index a17e184209..8c867b56eb 100644 --- a/src/llama-memory-hybrid-idx.h +++ b/src/llama-memory-hybrid-idx.h @@ -96,6 +96,11 @@ public: ple_history & ple_hist_get(llama_seq_id seq_id) const; private: + // forget seq_id (or, for seq_id < 0, everything) in every cache at once, so that a restore + // that failed partway cannot leave the indexer cache holding cells the attention cache does + // not. seq_id < 0 drops the whole context, as the caches themselves do on a failed restore. + void state_drop(llama_seq_id seq_id); + // the indexer cache holds one key head per layer, so it needs its own hparams: // llama_kv_cache keeps a reference to what it is given llama_hparams hparams_idx; From f43070aa5cc52cae7ddaefc43625357a6b88269d Mon Sep 17 00:00:00 2001 From: Daniel Han Date: Thu, 27 Aug 2026 04:13:14 +0000 Subject: [PATCH 2/9] kv-cache: clear the cache once when restoring a whole context state_read walks the streams of the cache in turn, and for a whole-context restore each stream went through state_read_meta, which starts by calling clear(). clear() resets every stream at once, so each stream after the first threw away the streams already restored, and the K/V buffers with them. A non-unified cache holds one stream per sequence, so a context saved with N sequences in it came back with only the sequence in the last stream that carried any cells - the highest sequence id. A unified cache has one stream and never showed it. The cache is now emptied once, before the loop, which is what a whole-context restore means. A blob whose streams are all empty now empties the cache as well, where before it left the old contents in place. --- src/llama-kv-cache.cpp | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/src/llama-kv-cache.cpp b/src/llama-kv-cache.cpp index 3330b66262..0490de24a4 100644 --- a/src/llama-kv-cache.cpp +++ b/src/llama-kv-cache.cpp @@ -2071,6 +2071,13 @@ const slot_info_vec_t * sinfos_in) { throw std::runtime_error("n_stream mismatch"); } + // a whole-context restore replaces every stream, so the cache is emptied once here. clear() + // resets all streams at once, so doing this per stream below would throw away the streams + // already read and leave only the last one + if (seq_id == -1) { + clear(true); + } + for (uint32_t s = 0; s < n_stream; ++s) { uint32_t cell_count; io.read(&cell_count, sizeof(cell_count)); @@ -2345,8 +2352,6 @@ bool llama_kv_cache::state_read_meta(llama_io_read_i & io, uint32_t strm, uint32 return false; } - clear(true); - for (uint32_t i = 0; i < cell_count; ++i) { llama_pos pos; uint32_t n_seq_id; From d1fe507f9526fddbdc87760a74b1e5dc906395f1 Mon Sep 17 00:00:00 2001 From: Daniel Han Date: Thu, 27 Aug 2026 04:13:22 +0000 Subject: [PATCH 3/9] kv-cache: check the mirrored slot layout on a whole-context restore too state_read_meta only looked at the layout it was given on the single-sequence path. A whole-context restore lays the cells out from 0 in both caches, so they agree as long as they restore the same number of cells, but nothing checked that they did: an indexer section belonging to some other context was read over cells the attention cache had filled from a different one, which is the state the indexer must never be left in. --- src/llama-kv-cache.cpp | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/src/llama-kv-cache.cpp b/src/llama-kv-cache.cpp index 0490de24a4..751bbe4e26 100644 --- a/src/llama-kv-cache.cpp +++ b/src/llama-kv-cache.cpp @@ -2352,6 +2352,14 @@ bool llama_kv_cache::state_read_meta(llama_io_read_i & io, uint32_t strm, uint32 return false; } + // the cells go in from 0, so a mirrored cache lands on the same ones as long as it + // restores the same count. the layout itself carries no more information here + if (sinfo_in && (sinfo_in->empty() || sinfo_in->n_stream() != 1 || sinfo_in->idxs[0].size() != cell_count)) { + LLAMA_LOG_ERROR("%s: mirrored slot layout holds %d cells, this cache restores %d\n", __func__, + sinfo_in->empty() ? 0 : (int) sinfo_in->idxs[0].size(), cell_count); + return false; + } + for (uint32_t i = 0; i < cell_count; ++i) { llama_pos pos; uint32_t n_seq_id; From bf5396021efd91b2a2253aa620a270e87d8955b5 Mon Sep 17 00:00:00 2001 From: Daniel Han Date: Thu, 27 Aug 2026 04:42:36 +0000 Subject: [PATCH 4/9] qwen4exp: give the PLE conv history its own mirrored recurrent row n_embd_r() reserved n_conv + ple_conv_state() so that one cache_r_l row could carry both the delta-net conv state and the PLE dilated conv history, but the QWEN4EXP arm of get_split_segments only described n_conv. Under -sm tensor the segment sum came up short by ple_conv_state() and llama_memory_recurrent construction aborted in ggml_backend_meta_alloc_ctx_tensors_from_buft. Widening the segment list is not the fix. The Meta backend propagates a view's split descriptor from its parent unchanged, so a view of one sub-range of a split axis is sized as the whole row on every device; declaring the PLE tail as a second segment merely moves the abort to "shape mismatch for VIEW" at graph allocation. The two histories also want opposite policies: the delta-net state is split by head to match wqkv and ssm_conv1d, while per_layer_tok_embd, ple_conv1d and ple_norm_conv are all mirrored, so every device computes the whole dilated conv and needs the whole history. One tensor cannot be both, and the split state has no per-segment mirroring. Move the PLE history into its own cache_ple_r_l%d row, mark it MIRRORED, and return n_embd_r() to n_conv. The row is allocated only on layers where is_ple holds, so mirroring one 92160-element row per device replaces a 92160-element tail on all 36 recurrent rows: the recurrent R footprint drops rather than grows. build_conv_state_at now takes its width from the tensor it was handed and keys its gather on that tensor, which also drops a cont of a strided view. --- src/llama-hparams.cpp | 8 +++-- src/llama-memory-recurrent.cpp | 61 +++++++++++++++++++++++++++++++--- src/llama-memory-recurrent.h | 4 +++ src/llama-model.cpp | 7 ++++ src/models/models.h | 7 ++-- src/models/qwen4exp.cpp | 32 ++++++++---------- 6 files changed, 89 insertions(+), 30 deletions(-) diff --git a/src/llama-hparams.cpp b/src/llama-hparams.cpp index a7d3106c04..6515ec3d14 100644 --- a/src/llama-hparams.cpp +++ b/src/llama-hparams.cpp @@ -203,9 +203,11 @@ uint32_t llama_hparams::n_embd_r() const { // Corresponds to Mamba's conv_states size const uint32_t n_conv = (ssm_d_conv > 0 ? ssm_d_conv - 1 : 0) * (ssm_d_inner + 2*ssm_n_group*ssm_d_state); - // qwen4exp puts a PLE module on a delta-net layer, so the row holds a second dilated conv - // state; the rows are uniform, so every recurrent layer reserves it - return n_conv + ple_conv_state(); + // qwen4exp's PLE dilated conv history deliberately does not share this row: the Meta backend + // splits cache_r_l by head and cannot view one sub-range of a split axis, so a second history + // packed behind the first is unaddressable under -sm tensor. it lives in cache_ple_r_l instead, + // mirrored, because the whole PLE module is mirrored + return n_conv; } uint32_t llama_hparams::n_embd_s() const { diff --git a/src/llama-memory-recurrent.cpp b/src/llama-memory-recurrent.cpp index e2990972ef..4e77ad0459 100644 --- a/src/llama-memory-recurrent.cpp +++ b/src/llama-memory-recurrent.cpp @@ -51,7 +51,8 @@ llama_memory_recurrent::llama_memory_recurrent( auto it = ctx_map.find(buft); if (it == ctx_map.end()) { ggml_init_params params = { - /*.mem_size =*/ size_t(2u*n_layer*ggml_tensor_overhead()), + // r and s per layer, plus the separate PLE conv row where the model has one + /*.mem_size =*/ size_t((hparams.ple_conv_state() > 0 ? 3u : 2u)*n_layer*ggml_tensor_overhead()), /*.mem_buffer =*/ NULL, /*.no_alloc =*/ true, }; @@ -71,6 +72,7 @@ llama_memory_recurrent::llama_memory_recurrent( r_l.resize(n_layer); s_l.resize(n_layer); + p_l.resize(n_layer); for (int i = 0; i < n_layer; i++) { if (filter && !filter(i)) { @@ -103,6 +105,14 @@ llama_memory_recurrent::llama_memory_recurrent( ggml_format_name(s, "cache_s_l%d", i); r_l[i] = r; s_l[i] = s; + + // qwen4exp's PLE history needs a row of its own so that the Meta backend can mirror it while + // the delta-net conv state next door stays split across devices + if (hparams.ple_conv_state() > 0 && hparams.is_ple(i)) { + ggml_tensor * p = ggml_new_tensor_2d(ctx, type_r, hparams.ple_conv_state(), n_rows); + ggml_format_name(p, "cache_ple_r_l%d", i); + p_l[i] = p; + } } // allocate tensors and initialize the buffers to avoid NaNs in the padding @@ -119,11 +129,13 @@ llama_memory_recurrent::llama_memory_recurrent( { const size_t memory_size_r = size_r_bytes(); const size_t memory_size_s = size_s_bytes(); + const size_t memory_size_p = size_p_bytes(); - LLAMA_LOG_INFO("%s: size = %7.2f MiB (%6u cells, %3d layers, %2u seqs %2u rs_seq), R (%s): %7.2f MiB, S (%s): %7.2f MiB\n", __func__, - (float)(memory_size_r + memory_size_s) / (1024.0f * 1024.0f), mem_size, n_layer, n_seq_max, n_rs_seq, + LLAMA_LOG_INFO("%s: size = %7.2f MiB (%6u cells, %3d layers, %2u seqs %2u rs_seq), R (%s): %7.2f MiB, S (%s): %7.2f MiB, P (%s): %7.2f MiB\n", __func__, + (float)(memory_size_r + memory_size_s + memory_size_p) / (1024.0f * 1024.0f), mem_size, n_layer, n_seq_max, n_rs_seq, ggml_type_name(type_r), (float)memory_size_r / (1024.0f * 1024.0f), - ggml_type_name(type_s), (float)memory_size_s / (1024.0f * 1024.0f)); + ggml_type_name(type_s), (float)memory_size_s / (1024.0f * 1024.0f), + ggml_type_name(type_r), (float)memory_size_p / (1024.0f * 1024.0f)); } } @@ -740,6 +752,18 @@ size_t llama_memory_recurrent::size_s_bytes() const { return size_s_bytes; } +size_t llama_memory_recurrent::size_p_bytes() const { + size_t size_p_bytes = 0; + + for (const auto & p : p_l) { + if (p != nullptr) { + size_p_bytes += ggml_nbytes(p); + } + } + + return size_p_bytes; +} + void llama_memory_recurrent::state_write(llama_io_write_i & io, llama_seq_id seq_id, llama_state_seq_flags flags) const { GGML_UNUSED(flags); @@ -899,6 +923,17 @@ void llama_memory_recurrent::state_write_data(llama_io_write_i & io, const std:: const size_t buf_size = range_size * r_size_row; io.write_tensor(r_l[il], range.first * r_size_row, buf_size); } + + // the PLE conv history is a second recurrent row, so it has to travel with the first + if (p_l[il] != nullptr) { + const uint64_t p_size_row = ggml_row_size(p_l[il]->type, hparams.ple_conv_state()); + io.write(&p_size_row, sizeof(p_size_row)); + + for (const auto & range : cell_ranges) { + const size_t range_size = range.second - range.first; + io.write_tensor(p_l[il], range.first * p_size_row, range_size * p_size_row); + } + } } if (!s_trans) { @@ -1097,6 +1132,20 @@ bool llama_memory_recurrent::state_read_data(llama_io_read_i & io, uint32_t cell // Read and set the keys for the whole cell range io.read_tensor(r_l[il], head * r_size_row, cell_count * r_size_row); } + + if (p_l[il] != nullptr) { + uint64_t p_size_row_ref; + io.read(&p_size_row_ref, sizeof(p_size_row_ref)); + const size_t p_size_row = ggml_row_size(p_l[il]->type, hparams.ple_conv_state()); + if (p_size_row != p_size_row_ref) { + LLAMA_LOG_ERROR("%s: mismatched ple row size (%zu != %zu, layer %d)\n", __func__, p_size_row, (size_t) p_size_row_ref, il); + return false; + } + + if (cell_count) { + io.read_tensor(p_l[il], head * p_size_row, cell_count * p_size_row); + } + } } if (!s_trans) { @@ -1251,6 +1300,10 @@ ggml_tensor * llama_memory_recurrent_context::get_s_l(int32_t il) const { return mem->s_l[il]; } +ggml_tensor * llama_memory_recurrent_context::get_p_l(int32_t il) const { + return mem->p_l[il]; +} + int32_t llama_memory_recurrent_context::s_copy(int i) const { const uint32_t cell_idx = i + mem->head; const int32_t src0 = mem->cells[cell_idx].src0; diff --git a/src/llama-memory-recurrent.h b/src/llama-memory-recurrent.h index b13b7b748f..4abb3f5cf5 100644 --- a/src/llama-memory-recurrent.h +++ b/src/llama-memory-recurrent.h @@ -111,6 +111,8 @@ public: // per layer std::vector r_l; std::vector s_l; + // a second conv history that must stay replicated across devices, so it cannot share the r row + std::vector p_l; private: //const llama_model & model; @@ -125,6 +127,7 @@ private: size_t size_r_bytes() const; size_t size_s_bytes() const; + size_t size_p_bytes() const; void state_write_meta(llama_io_write_i & io, const std::vector> & cell_ranges, llama_seq_id seq_id = -1) const; void state_write_data(llama_io_write_i & io, const std::vector> & cell_ranges) const; @@ -170,6 +173,7 @@ public: ggml_tensor * get_r_l(int32_t il) const; ggml_tensor * get_s_l(int32_t il) const; + ggml_tensor * get_p_l(int32_t il) const; int32_t s_copy(int i) const; diff --git a/src/llama-model.cpp b/src/llama-model.cpp index a926d32436..f8b4a30c97 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -395,6 +395,7 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str static const std::regex pattern_ssm_beta ("blk\\.\\d*\\.ssm_beta.weight"); static const std::regex pattern_ssm_beta_alpha ("blk\\.\\d*\\.ssm_ba.weight"); static const std::regex pattern_r_cache ("cache_r_l\\d*"); + static const std::regex pattern_ple_r_cache ("cache_ple_r_l\\d*"); static const std::regex pattern_s_cache ("cache_s_l\\d*"); static const std::regex pattern_ssm_conv1d ("blk\\.\\d*\\.ssm_conv1d.weight"); static const std::regex pattern_ssm_out_weight ("blk\\.\\d*\\.ssm_out.weight"); @@ -497,6 +498,12 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str return get_tensor_config_impl(GGML_BACKEND_SPLIT_AXIS_MIRRORED); } + // the PLE table is a model-level lookup and its conv kernel and norm are mirrored, so every + // device computes the whole dilated conv and needs the whole history + if (std::regex_match(tensor_name, pattern_ple_r_cache)) { + return get_tensor_config_impl(GGML_BACKEND_SPLIT_AXIS_MIRRORED); + } + // standard attention if (std::regex_match(tensor_name, pattern_q_weight) || std::regex_match(tensor_name, pattern_kv_weight)) { return get_tensor_config_impl(GGML_BACKEND_SPLIT_AXIS_1, "attn_output.weight", "ssm_out.weight"); diff --git a/src/models/models.h b/src/models/models.h index b22367fb0d..dacca61600 100644 --- a/src/models/models.h +++ b/src/models/models.h @@ -2341,17 +2341,16 @@ struct llama_model_qwen4exp : public llama_model_base { ggml_tensor * gate, int layer); - // build_rs writes the state tensor in place, so both convolutions share one gather per layer - std::map rs_rows; + // build_rs writes the state tensor in place, so one gather per cache tensor is reused + std::map rs_rows; - // conv history at an explicit offset: delta-net and PLE share the row + // one conv history per cache tensor: delta-net and PLE each have their own ggml_tensor * build_conv_state_at( llm_graph_input_rs * inp, ggml_tensor * conv_states_all, ggml_tensor * x, int64_t state_cols, int64_t channels, - int64_t row_offset, int il); ggml_tensor * build_ple( diff --git a/src/models/qwen4exp.cpp b/src/models/qwen4exp.cpp index c438710995..b49cf591a2 100644 --- a/src/models/qwen4exp.cpp +++ b/src/models/qwen4exp.cpp @@ -730,9 +730,8 @@ ggml_tensor * llama_model_qwen4exp::graph::build_layer_attn_linear( // the channels must match how load_arch_tensors sizes wqkv, not ssm_d_inner const int64_t conv_channels = head_k_dim * num_k_heads * 2 + head_v_dim * num_v_heads; - // offset 0: delta-net history first, PLE history (if any) after it ggml_tensor * conv_input = build_conv_state_at(inp, conv_states_all, qkv_mixed, - conv_kernel_size - 1, conv_channels, 0, il); + conv_kernel_size - 1, conv_channels, il); ggml_tensor * state = build_rs(inp, ssm_states_all, hparams.n_embd_s(), n_seqs); state = ggml_reshape_4d(ctx0, state, head_v_dim, head_v_dim, num_v_heads, n_seqs); @@ -985,36 +984,32 @@ void llm_graph_input_ple::set_input(const llama_ubatch * ubatch) { ggml_backend_tensor_set(rows, idx.data(), 0, idx.size()*ggml_element_size(rows)); } -// Read one conv history from the recurrent row at row_offset and write the new tail back. -// The shared build_conv_state cannot do this: the row holds the delta-net history and the PLE one. +// Read a conv history out of its own recurrent row and write the new tail back. +// The shared build_conv_state cannot do this: qwen4exp has two such rows per layer. ggml_tensor * llama_model_qwen4exp::graph::build_conv_state_at( llm_graph_input_rs * inp, ggml_tensor * conv_states_all, ggml_tensor * x, int64_t state_cols, int64_t channels, - int64_t row_offset, int il) { const auto * mctx_cur = inp->mctx; const auto kv_head = mctx_cur->get_head(); const int64_t n_seqs = ubatch.n_seqs; - const int64_t row_total = hparams.n_embd_r(); + const int64_t row_total = conv_states_all->ne[0]; - // the gather needs the whole row, then this convolution takes its slice - auto it = rs_rows.find(il); + // the row is exactly this convolution's state, so the gather is reused as a whole + GGML_ASSERT(state_cols * channels == row_total); + + auto it = rs_rows.find(conv_states_all); if (it == rs_rows.end()) { - it = rs_rows.emplace(il, build_rs(inp, conv_states_all, row_total, n_seqs)).first; + it = rs_rows.emplace(conv_states_all, build_rs(inp, conv_states_all, row_total, n_seqs)).first; } ggml_tensor * rows = it->second; - const size_t esz = ggml_element_size(rows); - - ggml_tensor * state = ggml_cont(ctx0, - ggml_view_2d(ctx0, rows, state_cols * channels, n_seqs, - rows->nb[1], row_offset * esz)); - state = ggml_reshape_3d(ctx0, state, state_cols, channels, n_seqs); + ggml_tensor * state = ggml_reshape_3d(ctx0, rows, state_cols, channels, n_seqs); cb(state, "conv_state_at", il); ggml_tensor * conv_input = ggml_concat(ctx0, state, ggml_transpose(ctx0, x), 0); @@ -1030,7 +1025,7 @@ ggml_tensor * llama_model_qwen4exp::graph::build_conv_state_at( ggml_tensor * dst = ggml_view_2d(ctx0, conv_states_all, state_cols * channels, n_seqs, conv_states_all->nb[1], - kv_head * row_size + row_offset * ggml_element_size(conv_states_all)); + kv_head * row_size); ggml_build_forward_expand(gf, ggml_cpy(ctx0, ggml_cont(ctx0, tail), dst)); @@ -1109,10 +1104,9 @@ ggml_tensor * llama_model_qwen4exp::graph::build_ple( const int64_t n_seq_tokens = ubatch.n_seq_tokens; // [hist + n_seq_tokens, hc_dim, n_seqs], tokens on ne[0] - ggml_tensor * padded = build_conv_state_at(inp, inp->mctx->get_r_l(il), + ggml_tensor * padded = build_conv_state_at(inp, inp->mctx->get_p_l(il), ggml_reshape_3d(ctx0, normalized, hc_dim, n_seq_tokens, n_seqs), - hist, hc_dim, - hparams.n_embd_r() - hparams.ple_conv_state(), il); + hist, hc_dim, il); ggml_tensor * conv_out = nullptr; for (int64_t k = 0; k < kern; ++k) { From 1caf6c5711ee3204b223f2e6197b8336eb6ef331 Mon Sep 17 00:00:00 2001 From: Daniel Han Date: Thu, 27 Aug 2026 06:16:54 +0000 Subject: [PATCH 5/9] llama: give the qwen4exp full memory context its indexer cache graph_reserve() walks a full memory context, and qwen4exp builds its sparse attention only when the context exposes an indexer cache. the full-context constructor left ctx_idx null, so the reserved worst case was the dense fallback: a smaller graph than the one decode executes. ggml-alloc then had to grow the compute buffer on the first decode, past the size reported at load. with -np 4 -c 32768 -fa on -ctk q8_0 -ctv q8_0 on an IQ1_S qwen4exp, the reserved CUDA0 buffer was 217.00 MiB against 275.71 MiB actually used, and CUDA_Host 42.31 MiB against 191.14 MiB. reserving the sparse graph makes both match exactly, in unified and non-unified cache mode. Co-authored-by: Pascal Assisted-by: Claude --- src/llama-memory-hybrid-idx.cpp | 9 ++++++++- src/llama-memory-hybrid-idx.h | 4 ++-- 2 files changed, 10 insertions(+), 3 deletions(-) diff --git a/src/llama-memory-hybrid-idx.cpp b/src/llama-memory-hybrid-idx.cpp index 38e3e61c57..5cef7a032e 100644 --- a/src/llama-memory-hybrid-idx.cpp +++ b/src/llama-memory-hybrid-idx.cpp @@ -557,7 +557,14 @@ llama_memory_hybrid_idx_context::llama_memory_hybrid_idx_context(llama_memory_st llama_memory_hybrid_idx_context::llama_memory_hybrid_idx_context(llama_memory_hybrid_idx * mem) : llama_memory_hybrid_context(mem), - mem(mem) {} + mem(mem), + // graph reservation walks a full context, and qwen4exp builds the sparse attention only when + // this is set. without it the reserved worst case is the smaller dense graph, so ggml-alloc + // must grow the compute buffer on the first decode + ns_ubatch(mem->get_mem_idx() == nullptr ? + std::vector() : std::vector{ mem->get_mem_idx()->get_n_stream() }), + ctx_idx(mem->get_mem_idx() == nullptr ? nullptr : + new llama_kv_cache_context(mem->get_mem_idx())) {} llama_memory_hybrid_idx_context::llama_memory_hybrid_idx_context( llama_memory_hybrid_idx * mem, diff --git a/src/llama-memory-hybrid-idx.h b/src/llama-memory-hybrid-idx.h index 8c867b56eb..c16fe4eeb0 100644 --- a/src/llama-memory-hybrid-idx.h +++ b/src/llama-memory-hybrid-idx.h @@ -157,7 +157,7 @@ public: // llama_memory_hybrid_idx_context specific API // - // nullptr with no indexer, and for the full and update contexts, which build no sparse graph + // nullptr with no indexer, and for the update context, which builds no sparse graph const llama_kv_cache_context * get_idx() const; // streams in the current slot info, the `ns` of get_k/get_v; 1 if unified @@ -182,7 +182,7 @@ private: // declared first, so it is initialised while sinfos_idx is still intact const std::vector ns_ubatch; - // null unless the model has an indexer and this is a batch context + // null unless the model has an indexer and this is a batch or full context const llama_memory_context_ptr ctx_idx; // mirrors the base class's ubatch cursor, which is private there From c3259772c711c98da4144bdd1e0fd70f303639ae Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Thu, 27 Aug 2026 07:32:16 +0000 Subject: [PATCH 6/9] qwen4exp: shrink the PLE hparams storage llama_hparams is held by value inside llm_graph_params and every llm_graph_input_*, and llm_graph_params is a stack local in graph_reserve and process_ubatch, so its width is paid on every worker thread stack. is_ple_impl spent 2048 bytes carrying 512 bits. It is the one per-layer flag that is not moved through the loader's uint32 array templates, so a bitset costs nothing in call sites and also removes the uninitialized read that non-qwen4exp archs had, since nothing filled the array for them. The PLE head offsets and vocab sizes are token-space indices; the gather that consumes them already truncates to int32, so 64-bit storage was never reachable. The gguf arrays stay uint64 for file compatibility and are narrowed on load. sizeof(llama_hparams) 34440 -> 31944, sizeof(llm_graph_params) 34872 -> 32376. --- src/llama-hparams.h | 13 ++++++++++--- src/models/qwen4exp.cpp | 19 +++++++++++++++---- 2 files changed, 25 insertions(+), 7 deletions(-) diff --git a/src/llama-hparams.h b/src/llama-hparams.h index d829d034cb..739156da6d 100644 --- a/src/llama-hparams.h +++ b/src/llama-hparams.h @@ -3,6 +3,7 @@ #include "llama.h" #include +#include #include #include @@ -283,10 +284,16 @@ struct llama_hparams { uint32_t ple_eos_token_id = 0; // the id the PLE hash stands in at image positions; 0 makes the loader fall back to EOS uint32_t ple_image_token_id = 0; - std::array is_ple_impl; + // unlike is_swa_impl and friends this is never read or written as a per-layer gguf array + // (the file lists PLE layer indices), so it is not tied to the loader's uint32 array type + // and can hold one bit per layer instead of one word + std::bitset is_ple_impl; + // the hash multipliers reach ~2e13 and have to stay 64-bit std::array ple_layer_multipliers; - std::array ple_head_offsets; - std::array ple_head_vocab_sizes; + // head offsets and vocab sizes are token-space indices; the gather that consumes them + // truncates to int32, so 64-bit storage could never have been used + std::array ple_head_offsets; + std::array ple_head_vocab_sizes; bool is_ple(uint32_t il) const; diff --git a/src/models/qwen4exp.cpp b/src/models/qwen4exp.cpp index b49cf591a2..533e9a300b 100644 --- a/src/models/qwen4exp.cpp +++ b/src/models/qwen4exp.cpp @@ -33,7 +33,7 @@ void llama_model_qwen4exp::load_arch_hparams(llama_model_loader & ml) { ml.get_key_or_arr(LLM_KV_ATTENTION_COMPRESS_RATIOS, hparams.dsv4_compress_ratios, hparams.n_layer_all, false); // PLE n-gram hash embeddings; if the key group is absent every field stays zero - std::fill(hparams.is_ple_impl.begin(), hparams.is_ple_impl.end(), 0); + hparams.is_ple_impl.reset(); hparams.ple_n_heads = 0; uint32_t n_ple = 0; @@ -43,7 +43,7 @@ void llama_model_qwen4exp::load_arch_hparams(llama_model_loader & ml) { ml.get_arr(LLM_KV_PLE_LAYERS, ple_layers); for (uint32_t il : ple_layers) { GGML_ASSERT(il < hparams.n_layer_all); - hparams.is_ple_impl[il] = 1; + hparams.is_ple_impl.set(il); } ml.get_key(LLM_KV_PLE_NGRAM_SIZE, hparams.ple_ngram_size); @@ -60,8 +60,19 @@ void llama_model_qwen4exp::load_arch_hparams(llama_model_loader & ml) { GGML_ASSERT(hparams.ple_n_heads > 0 && hparams.ple_n_heads <= LLAMA_MAX_PLE_HEADS); ml.get_arr(LLM_KV_PLE_LAYER_MULTIPLIERS, hparams.ple_layer_multipliers); - ml.get_arr(LLM_KV_PLE_HEAD_OFFSETS, hparams.ple_head_offsets); - ml.get_arr(LLM_KV_PLE_HEAD_VOCAB_SIZES, hparams.ple_head_vocab_sizes); + + // the file writes the head ranges as uint64 arrays, so read them at that width and + // narrow; hparams keeps them at the int32 width the row gather actually uses + std::array head_offsets = {}; + std::array head_vocab_sizes = {}; + ml.get_arr(LLM_KV_PLE_HEAD_OFFSETS, head_offsets); + ml.get_arr(LLM_KV_PLE_HEAD_VOCAB_SIZES, head_vocab_sizes); + for (uint32_t h = 0; h < hparams.ple_n_heads; ++h) { + GGML_ASSERT(head_offsets[h] + head_vocab_sizes[h] <= INT32_MAX && + "PLE head range does not fit the int32 row index"); + hparams.ple_head_offsets[h] = (uint32_t) head_offsets[h]; + hparams.ple_head_vocab_sizes[h] = (uint32_t) head_vocab_sizes[h]; + } } // linear attention everywhere except every full_attention_interval-th layer From 95da4ba86065dda8659981a508dfb52d8c3f1032 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Thu, 27 Aug 2026 04:20:48 +0000 Subject: [PATCH 7/9] llama: opt-in random-access mmap advice for host-resident gather tables qwen4exp keeps per_layer_token_embd on the host: 26.8 GiB at IQ4_NL, read by ggml_get_rows as 16 gathers of ~90-170 bytes per token, spread across 16 head regions ~20M rows apart. Measured over 4.75M gathers, no two consecutive gathers land on the same 4 KiB page, so the readahead the loader asks for buys nothing here and the whole table ends up cached to serve about 4% of itself. llama_mmap applies POSIX_FADV_SEQUENTIAL, MAP_POPULATE and a whole-file POSIX_MADV_WILLNEED unconditionally. Those are right for streaming the file once into buffers and wrong for whatever stays mapped afterwards. Under LLAMA_MMAP_RANDOM the eager pull-in is skipped and the mapping is advised random once every tensor has been read, so the load itself keeps its sequential readahead. That alone drops the table to 4.4% resident but serializes one NVMe latency per gather. The second half is what pays for it: the PLE input already computes every row index for the ubatch before the graph runs, so the pages those rows fall on are handed to the kernel in one batch and the reads overlap. POSIX_MADV_WILLNEED on POSIX, PrefetchVirtualMemory on Windows, which takes the discontiguous ranges in a single call. Off by default and off for every other model: the batched prefetch keys off "this mapping was advised random", which nothing sets unless the user opts in. -c 512 --chunks 60, cold, IQ1_S, mean of 3: default 35.3 s 26.82 GiB resident (100%) advice only 104.5 s 1.19 GiB resident (4.4%) advice + prefetch 34.2 s 1.19 GiB resident (4.4%) PPL 4.2346 +/- 0.07862 in all three. IQ1_S KLD is unchanged in every field, including Mean KLD 0.396070 +/- 0.001931 and Same top p 77.325%. --- src/llama-mmap.cpp | 230 ++++++++++++++++++++++++++++++++++++++++ src/llama-mmap.h | 30 ++++++ src/llama-model.cpp | 37 +++++++ src/llama-model.h | 8 ++ src/models/qwen4exp.cpp | 5 + 5 files changed, 310 insertions(+) diff --git a/src/llama-mmap.cpp b/src/llama-mmap.cpp index ed572da7fb..33167c7280 100644 --- a/src/llama-mmap.cpp +++ b/src/llama-mmap.cpp @@ -5,7 +5,9 @@ #include "ggml.h" #include +#include #include +#include #include #include #include @@ -438,6 +440,87 @@ void llama_file::write_u32(uint32_t val) const { pimpl->write_u32(val); } // llama_mmap +llama_mmap_random_mode llama_mmap_random_mode_get() { + // read once: this is consulted per mapping and per gather + static const llama_mmap_random_mode mode = []() { + const char * env = getenv("LLAMA_MMAP_RANDOM"); + if (env == nullptr || strcmp(env, "0") == 0 || env[0] == '\0') { + return LLAMA_MMAP_RANDOM_OFF; + } + if (strcmp(env, "drop") == 0) { + return LLAMA_MMAP_RANDOM_DROP; + } + return LLAMA_MMAP_RANDOM_ON; + }(); + + return mode; +} + +bool llama_mmap_random_prefetch_enabled() { + // on with the feature, so the batched readahead that pays for the random hints cannot be + // left off by accident. separate only so the two halves can be measured apart. + static const bool enabled = []() { + const char * env = getenv("LLAMA_MMAP_RANDOM_PREFETCH"); + return env == nullptr ? true : strcmp(env, "0") != 0; + }(); + + return llama_mmap_random_mode_get() != LLAMA_MMAP_RANDOM_OFF && enabled; +} + +static size_t llama_mmap_page_size() { +#if defined(_WIN32) + SYSTEM_INFO si; + GetSystemInfo(&si); + return (size_t) si.dwPageSize; +#elif defined(_SC_PAGESIZE) + return (size_t) sysconf(_SC_PAGESIZE); +#else + return 4096; +#endif +} + +// the distinct pages the given rows fall on, as offsets into the mapping, merged into runs. +// a row is much smaller than a page and rows repeat within a batch, so this is what turns a +// hint per row into a hint per page. platform independent: the callers differ only in which +// syscall they hand the result to. +static std::vector> llama_mmap_row_pages( + size_t base_off, size_t stride, size_t row_size, size_t map_size, + const int32_t * rows, size_t n_rows, size_t page_size) { + std::vector pages; + pages.reserve(n_rows); + + for (size_t i = 0; i < n_rows; ++i) { + if (rows[i] < 0) { + continue; + } + const size_t first = base_off + (size_t) rows[i] * stride; + const size_t last = first + row_size; + // a corrupt or unexpected index must not turn into a hint outside the mapping + if (row_size == 0 || last > map_size || first < base_off) { + continue; + } + for (size_t p = first / page_size; p <= (last - 1) / page_size; ++p) { + pages.push_back(p); + } + } + + std::sort(pages.begin(), pages.end()); + pages.erase(std::unique(pages.begin(), pages.end()), pages.end()); + + std::vector> ranges; + for (size_t i = 0; i < pages.size(); ) { + size_t j = i + 1; + while (j < pages.size() && pages[j] == pages[j - 1] + 1) { + ++j; + } + const size_t off = pages[i] * page_size; + ranges.emplace_back(off, std::min((pages[j - 1] - pages[i] + 1) * page_size, map_size - off)); + i = j; + } + + return ranges; +} + struct llama_mmap::impl { #ifdef _POSIX_MAPPED_FILES std::vector> mapped_fragments; @@ -445,8 +528,13 @@ struct llama_mmap::impl { impl(struct llama_file * file, size_t prefetch, bool numa) { size = file->size(); int fd = file->file_id(); + fd_advise = fd; int flags = MAP_SHARED; if (numa) { prefetch = 0; } + // with the random hints the eager pull-in is pure waste: it reads the whole file to + // populate pages that the gathers will then hit at most a few percent of. the loader + // still gets sequential readahead for the tensors it actually copies out. + if (llama_mmap_random_mode_get() != LLAMA_MMAP_RANDOM_OFF) { prefetch = 0; } #ifdef __linux__ if (posix_fadvise(fd, 0, 0, POSIX_FADV_SEQUENTIAL)) { LLAMA_LOG_WARN("warning: posix_fadvise(.., POSIX_FADV_SEQUENTIAL) failed: %s\n", @@ -475,6 +563,54 @@ struct llama_mmap::impl { mapped_fragments.emplace_back(0, file->size()); } + // the load path asks for POSIX_FADV_SEQUENTIAL, which is right while the file is being + // streamed once into buffers and wrong for whatever stays host-resident afterwards: those + // tensors are read by sparse gathers, where readahead and fault-around buy nothing and cost + // page cache. flipping the advice only after load keeps both halves happy. + void advise_random(bool drop) { +#if defined(__linux__) + if (drop) { + // hand back what the load pulled in; the hot pages fault back on demand + if (madvise(addr, size, MADV_DONTNEED)) { + LLAMA_LOG_WARN("warning: madvise(.., MADV_DONTNEED) failed: %s\n", strerror(errno)); + } + if (fd_advise >= 0 && posix_fadvise(fd_advise, 0, 0, POSIX_FADV_DONTNEED)) { + LLAMA_LOG_WARN("warning: posix_fadvise(.., POSIX_FADV_DONTNEED) failed: %s\n", strerror(errno)); + } + } + if (fd_advise >= 0 && posix_fadvise(fd_advise, 0, 0, POSIX_FADV_RANDOM)) { + LLAMA_LOG_WARN("warning: posix_fadvise(.., POSIX_FADV_RANDOM) failed: %s\n", strerror(errno)); + } +#else + GGML_UNUSED(drop); +#endif +#if defined(_POSIX_MAPPED_FILES) + if (posix_madvise(addr, size, POSIX_MADV_RANDOM)) { + LLAMA_LOG_WARN("warning: posix_madvise(.., POSIX_MADV_RANDOM) failed: %s\n", strerror(errno)); + } +#endif + } + + void prefetch_rows(const void * base, size_t stride, size_t row_size, + const int32_t * rows, size_t n_rows) const { +#if defined(_POSIX_MAPPED_FILES) + const size_t base_off = (const char *) base - (const char *) addr; + + for (const auto & [off, len] : llama_mmap_row_pages( + base_off, stride, row_size, size, rows, n_rows, llama_mmap_page_size())) { + // deliberately unchecked: this is a hint issued thousands of times per batch, and a + // failed hint only costs the fault it would have avoided + posix_madvise((char *) addr + off, len, POSIX_MADV_WILLNEED); + } +#else + GGML_UNUSED(base); + GGML_UNUSED(stride); + GGML_UNUSED(row_size); + GGML_UNUSED(rows); + GGML_UNUSED(n_rows); +#endif + } + static void align_range(size_t * first, size_t * last, size_t page_size) { size_t offset_in_page = *first & (page_size - 1); size_t offset_to_page = offset_in_page == 0 ? 0 : page_size - offset_in_page; @@ -547,6 +683,9 @@ struct llama_mmap::impl { throw std::runtime_error(format("CreateFileMappingA failed: %s", llama_format_win_err(error).c_str())); } + // see the POSIX branch: opting in means the eager pull-in is waste + if (llama_mmap_random_mode_get() != LLAMA_MMAP_RANDOM_OFF) { prefetch = 0; } + addr = MapViewOfFile(hMapping, FILE_MAP_READ, 0, 0, 0); DWORD error = GetLastError(); @@ -582,6 +721,50 @@ struct llama_mmap::impl { GGML_UNUSED(last); } + // Windows has no per-mapping "read this randomly" hint. skipping the eager + // PrefetchVirtualMemory in the constructor is what keeps the pages out; there is nothing + // further to say here, and nothing to drop back. + void advise_random(bool drop) { + GGML_UNUSED(drop); + } + + // PrefetchVirtualMemory takes the whole set of ranges in one call, which is exactly the + // batching this wants: the reads are issued together instead of one fault at a time. + void prefetch_rows(const void * base, size_t stride, size_t row_size, + const int32_t * rows, size_t n_rows) const { +#if _WIN32_WINNT >= 0x602 + BOOL (WINAPI *pPrefetchVirtualMemory) (HANDLE, ULONG_PTR, PWIN32_MEMORY_RANGE_ENTRY, ULONG); + HMODULE hKernel32 = GetModuleHandleW(L"kernel32.dll"); + + pPrefetchVirtualMemory = (decltype(pPrefetchVirtualMemory))(void *) GetProcAddress(hKernel32, "PrefetchVirtualMemory"); + if (!pPrefetchVirtualMemory) { + return; + } + + const size_t base_off = (const char *) base - (const char *) addr; + + std::vector entries; + for (const auto & [off, len] : llama_mmap_row_pages( + base_off, stride, row_size, size, rows, n_rows, llama_mmap_page_size())) { + WIN32_MEMORY_RANGE_ENTRY e; + e.VirtualAddress = (char *) addr + off; + e.NumberOfBytes = (SIZE_T) len; + entries.push_back(e); + } + + if (!entries.empty()) { + // unchecked for the same reason as the POSIX branch: it is only a hint + pPrefetchVirtualMemory(GetCurrentProcess(), (ULONG_PTR) entries.size(), entries.data(), 0); + } +#else + GGML_UNUSED(base); + GGML_UNUSED(stride); + GGML_UNUSED(row_size); + GGML_UNUSED(rows); + GGML_UNUSED(n_rows); +#endif + } + ~impl() { if (hMapping) { if (addr) { @@ -611,10 +794,38 @@ struct llama_mmap::impl { throw std::runtime_error("mmap not supported"); } + + void advise_random(bool drop) { + GGML_UNUSED(drop); + + throw std::runtime_error("mmap not supported"); + } + + void prefetch_rows(const void * base, size_t stride, size_t row_size, + const int32_t * rows, size_t n_rows) const { + GGML_UNUSED(base); + GGML_UNUSED(stride); + GGML_UNUSED(row_size); + GGML_UNUSED(rows); + GGML_UNUSED(n_rows); + + throw std::runtime_error("mmap not supported"); + } #endif + bool contains(const void * ptr, size_t len) const { + const char * p = (const char *) ptr; + const char * b = (const char *) addr; + + return p >= b && len <= size && (size_t) (p - b) <= size - len; + } + void * addr; size_t size; + + // the fd is kept only to re-advise the file; the mapping owns no reference to it + int fd_advise = -1; + bool random = false; }; llama_mmap::llama_mmap(struct llama_file * file, size_t prefetch, bool numa) : pimpl(std::make_unique(file, prefetch, numa)) {} @@ -625,6 +836,25 @@ void * llama_mmap::addr() const { return pimpl->addr; } void llama_mmap::unmap_fragment(size_t first, size_t last) { pimpl->unmap_fragment(first, last); } +void llama_mmap::advise_random(bool drop) { + pimpl->advise_random(drop); + // set here rather than per platform: the flag means "the caller opted this mapping in", and + // the batched prefetch is worth doing even where the advice itself is a no-op + pimpl->random = true; +} + +bool llama_mmap::is_random() const { return pimpl->random; } + +bool llama_mmap::contains(const void * ptr, size_t len) const { return pimpl->contains(ptr, len); } + +void llama_mmap::prefetch_rows(const void * base, size_t stride, size_t row_size, + const int32_t * rows, size_t n_rows) const { + if (!pimpl->random) { + return; + } + pimpl->prefetch_rows(base, stride, row_size, rows, n_rows); +} + #if defined(_POSIX_MEMLOCK_RANGE) || defined(_WIN32) const bool llama_mmap::SUPPORTED = true; #else diff --git a/src/llama-mmap.h b/src/llama-mmap.h index b7d5c61e95..ab4ef60408 100644 --- a/src/llama-mmap.h +++ b/src/llama-mmap.h @@ -50,6 +50,22 @@ struct llama_mmap { void unmap_fragment(size_t first, size_t last); + // opt-in, see llama_mmap_random_mode(). marks the whole mapping as randomly accessed, which + // is only correct once loading is done: until then the loader streams the file sequentially. + void advise_random(bool drop); + + // true once advise_random() has been applied. everything that keys off "this mapping is read + // randomly" tests this, so nothing changes for a mapping the user did not opt in for. + bool is_random() const; + + // true if [ptr, ptr + len) lies inside this mapping + bool contains(const void * ptr, size_t len) const; + + // ask the kernel to start reading the given rows. issued as one batch so the faults overlap + // instead of serializing; a no-op unless the mapping is marked random. + void prefetch_rows(const void * base, size_t stride, size_t row_size, + const int32_t * rows, size_t n_rows) const; + static const bool SUPPORTED; private: @@ -57,6 +73,20 @@ private: std::unique_ptr pimpl; }; +// how the model file mappings should be advised, from the LLAMA_MMAP_RANDOM environment variable. +// off unless the user asks: the random hints cost a large cold-prefill slowdown on models whose +// host-resident tensors are read sequentially, so this cannot be a default. +enum llama_mmap_random_mode { + LLAMA_MMAP_RANDOM_OFF = 0, // upstream behaviour + LLAMA_MMAP_RANDOM_ON = 1, // skip MAP_POPULATE/WILLNEED, advise random after load + LLAMA_MMAP_RANDOM_DROP = 2, // additionally drop what the load pulled in +}; + +llama_mmap_random_mode llama_mmap_random_mode_get(); + +// whether batched readahead ahead of a sparse gather is enabled (LLAMA_MMAP_RANDOM_PREFETCH) +bool llama_mmap_random_prefetch_enabled(); + struct llama_mlock { llama_mlock(); ~llama_mlock(); diff --git a/src/llama-model.cpp b/src/llama-model.cpp index f8b4a30c97..925e2aca61 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -1149,6 +1149,10 @@ struct llama_model::impl { // model memory mapped files llama_mmaps mappings; + // set once loading is done, if any mapping was advised random. lets prefetch_rows() bail out + // without walking the mappings, which is the only cost the feature has when it is off. + bool mappings_random = false; + // objects representing data potentially being locked in memory llama_mlocks mlock_bufs; llama_mlocks mlock_mmaps; @@ -1812,11 +1816,44 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) { for (auto & mapping : ml.mappings) { pimpl->mappings.emplace_back(std::move(mapping)); } + + // only now that every tensor has been read is it safe to say the file is read randomly: + // the load itself is a sequential pass and wants the readahead it has been getting. + const llama_mmap_random_mode random_mode = llama_mmap_random_mode_get(); + if (random_mode != LLAMA_MMAP_RANDOM_OFF) { + for (auto & mapping : pimpl->mappings) { + mapping->advise_random(random_mode == LLAMA_MMAP_RANDOM_DROP); + } + pimpl->mappings_random = !pimpl->mappings.empty(); + + LLAMA_LOG_INFO("%s: LLAMA_MMAP_RANDOM: advised %zu mapping(s) for random access%s\n", + __func__, pimpl->mappings.size(), + random_mode == LLAMA_MMAP_RANDOM_DROP ? ", dropped cached pages" : ""); + } } return true; } +void llama_model::prefetch_rows(const struct ggml_tensor * t, const int32_t * rows, size_t n_rows) const { + if (!pimpl->mappings_random || t == nullptr || t->data == nullptr || n_rows == 0) { + return; + } + if (!llama_mmap_random_prefetch_enabled()) { + return; + } + + // rows are addressed off the tensor, so the whole tensor has to sit in the mapping we find + const size_t nbytes = ggml_nbytes(t); + + for (const auto & mapping : pimpl->mappings) { + if (mapping->is_random() && mapping->contains(t->data, nbytes)) { + mapping->prefetch_rows(t->data, t->nb[1], ggml_row_size(t->type, t->ne[0]), rows, n_rows); + return; + } + } +} + ggml_tensor * llama_model_base::create_tensor(llama_model_loader & ml, const LLM_TN_IMPL & tn, const std::initializer_list & ne, int flags) { const buft_list_t * buft_list_layer = tn.bid == -1 ? nullptr : pimpl->dev_layer.at(tn.bid).buft_list; return ml.create_tensor( diff --git a/src/llama-model.h b/src/llama-model.h index a2c25c6381..cf7c75a476 100644 --- a/src/llama-model.h +++ b/src/llama-model.h @@ -734,6 +734,14 @@ struct llama_model { const struct ggml_tensor * get_tensor(const char * name) const; + // ask the kernel to start reading the rows a gather is about to take out of a host-mapped + // tensor, so the faults overlap instead of serializing one NVMe latency at a time. + // + // does nothing unless the tensor lives in a mapping that was advised random, which only + // happens under LLAMA_MMAP_RANDOM. off, and for anything not read out of a mapping (offloaded + // tensors, --load-mode none, non-POSIX hosts), this is one bool test. + void prefetch_rows(const struct ggml_tensor * t, const int32_t * rows, size_t n_rows) const; + float get_rope_freq_base (const llama_cparams & cparams, int il) const; float get_rope_freq_scale(const llama_cparams & cparams, int il) const; diff --git a/src/models/qwen4exp.cpp b/src/models/qwen4exp.cpp index 533e9a300b..34282aa0c4 100644 --- a/src/models/qwen4exp.cpp +++ b/src/models/qwen4exp.cpp @@ -992,6 +992,11 @@ void llm_graph_input_ple::set_input(const llama_ubatch * ubatch) { h.next_pos = pos + 1; } + // the table is far too big to offload, so it is gathered straight out of the mapping: one + // fault per row, 16 per token, no two of them on the same page. left to the get_rows those + // faults happen one at a time; queued here they are in flight before the graph even runs. + pmodel.prefetch_rows(pmodel.per_layer_tok_embd, idx.data(), idx.size()); + ggml_backend_tensor_set(rows, idx.data(), 0, idx.size()*ggml_element_size(rows)); } From 58325573c9c553d97d30a06b291b7b49bef7b3d4 Mon Sep 17 00:00:00 2001 From: danielhanchen Date: Thu, 27 Aug 2026 06:06:07 +0000 Subject: [PATCH 8/9] llama: narrow the random-access mmap advice to the gather table The advice was applied per mapping: every mapping the model kept got POSIX_MADV_RANDOM plus a whole-file POSIX_FADV_RANDOM, and the eager pull-in was skipped for every file. On qwen4exp that also hit token_embd.weight, which sits 0.33 GiB past the PLE table in the same shard and is read densely, not by sparse gathers. Measured over -c 512 --chunks 60 on IQ1_S it fell to 8.45% resident, against 100% with the feature off. A model now nominates its gather tables (qwen4exp: per_layer_tok_embd) and only those byte ranges are advised. The range is rounded out to whole pages, which on this model takes in 832 bytes before and 192 after. token_embd goes back to 86.55% resident and the PLE table still drops to 4.44%; smaps shows one VM_RAND_READ VMA of exactly the table instead of one over all 27.16 GiB that stays mapped. posix_fadvise is dropped from the narrowed path. POSIX_FADV_RANDOM ignores its offset and length and marks the whole open file, and the FMODE_RANDOM it sets is only read by page_cache_sync_ra() on the read() path, which a fault on a MADV_RANDOM vma never reaches. POSIX_FADV_ DONTNEED does take a range, so the drop mode keeps it. The eager pull-in is now skipped only for the files holding a nominated table, and re-issued as WILLNEED over the rest of such a file, so other shards load exactly as before. prefetch_rows() keys off the tensor being nominated rather than off a mapping-level flag, so the batched readahead lands only where the advice did. -c 512 --chunks 60, cold, IQ1_S, mean of 3, total wall: default 32.50 s whole mapping 30.05 s narrowed 30.35 s PPL 4.2061 in all three. IQ1_S KLD is bit-identical with the feature on and off, including Mean KLD 0.396070 +/- 0.001931 and Same top p 77.325%. tg128 73.65 +/- 0.33 narrowed against 73.49 +/- 0.34 whole. Assisted-by: Claude --- src/llama-mmap.cpp | 141 ++++++++++++++++++++++++++++--------- src/llama-mmap.h | 19 ++--- src/llama-model-loader.cpp | 15 +++- src/llama-model-loader.h | 4 ++ src/llama-model.cpp | 64 ++++++++++++----- src/llama-model.h | 14 +++- src/models/models.h | 8 +++ 7 files changed, 199 insertions(+), 66 deletions(-) diff --git a/src/llama-mmap.cpp b/src/llama-mmap.cpp index 33167c7280..a615eb6ef9 100644 --- a/src/llama-mmap.cpp +++ b/src/llama-mmap.cpp @@ -531,10 +531,6 @@ struct llama_mmap::impl { fd_advise = fd; int flags = MAP_SHARED; if (numa) { prefetch = 0; } - // with the random hints the eager pull-in is pure waste: it reads the whole file to - // populate pages that the gathers will then hit at most a few percent of. the loader - // still gets sequential readahead for the tensors it actually copies out. - if (llama_mmap_random_mode_get() != LLAMA_MMAP_RANDOM_OFF) { prefetch = 0; } #ifdef __linux__ if (posix_fadvise(fd, 0, 0, POSIX_FADV_SEQUENTIAL)) { LLAMA_LOG_WARN("warning: posix_fadvise(.., POSIX_FADV_SEQUENTIAL) failed: %s\n", @@ -565,30 +561,63 @@ struct llama_mmap::impl { // the load path asks for POSIX_FADV_SEQUENTIAL, which is right while the file is being // streamed once into buffers and wrong for whatever stays host-resident afterwards: those - // tensors are read by sparse gathers, where readahead and fault-around buy nothing and cost - // page cache. flipping the advice only after load keeps both halves happy. - void advise_random(bool drop) { + // tensors are read by sparse gathers, where readahead buys nothing and costs page cache. + // flipping the advice only after load, and only over the tensor, keeps everything else on + // the loader's behaviour. + void advise_random_range(size_t offset, size_t len, bool drop) { + if (offset >= size || len == 0) { + return; + } + len = std::min(len, size - offset); + + // madvise rejects an unaligned start and rounds the length up, so round both out. that + // can take in the tail of the tensor before and the head of the one after, one page each + const size_t page = llama_mmap_page_size(); + const size_t first = offset & ~(page - 1); + const size_t last = std::min(size, (offset + len + page - 1) & ~(page - 1)); + #if defined(__linux__) if (drop) { - // hand back what the load pulled in; the hot pages fault back on demand - if (madvise(addr, size, MADV_DONTNEED)) { + // on a shared file map this only tears down our page tables + if (madvise((char *) addr + first, last - first, MADV_DONTNEED)) { LLAMA_LOG_WARN("warning: madvise(.., MADV_DONTNEED) failed: %s\n", strerror(errno)); } - if (fd_advise >= 0 && posix_fadvise(fd_advise, 0, 0, POSIX_FADV_DONTNEED)) { + // and this frees the page cache. it takes the range and spares partial pages, so a + // tensor sharing the first or last page keeps its cache + if (fd_advise >= 0 && posix_fadvise(fd_advise, (off_t) offset, (off_t) len, POSIX_FADV_DONTNEED)) { LLAMA_LOG_WARN("warning: posix_fadvise(.., POSIX_FADV_DONTNEED) failed: %s\n", strerror(errno)); } } - if (fd_advise >= 0 && posix_fadvise(fd_advise, 0, 0, POSIX_FADV_RANDOM)) { - LLAMA_LOG_WARN("warning: posix_fadvise(.., POSIX_FADV_RANDOM) failed: %s\n", strerror(errno)); - } #else GGML_UNUSED(drop); #endif -#if defined(_POSIX_MAPPED_FILES) - if (posix_madvise(addr, size, POSIX_MADV_RANDOM)) { + // no POSIX_FADV_RANDOM to go with this: it ignores the range and marks the whole open + // file, and the FMODE_RANDOM it sets is only read by the read() path, never by a fault + if (posix_madvise((char *) addr + first, last - first, POSIX_MADV_RANDOM)) { LLAMA_LOG_WARN("warning: posix_madvise(.., POSIX_MADV_RANDOM) failed: %s\n", strerror(errno)); } -#endif + } + + void prefetch_except(const std::vector> & skip) { + const size_t page = llama_mmap_page_size(); + + size_t pos = 0; + for (const auto & [off, len] : skip) { + const size_t first = off & ~(page - 1); + if (first > pos) { + prefetch_range(pos, first - pos); + } + pos = std::max(pos, std::min(size, (off + len + page - 1) & ~(page - 1))); + } + if (pos < size) { + prefetch_range(pos, size - pos); + } + } + + void prefetch_range(size_t offset, size_t len) const { + if (posix_madvise((char *) addr + offset, len, POSIX_MADV_WILLNEED)) { + LLAMA_LOG_WARN("warning: posix_madvise(.., POSIX_MADV_WILLNEED) failed: %s\n", strerror(errno)); + } } void prefetch_rows(const void * base, size_t stride, size_t row_size, @@ -683,9 +712,6 @@ struct llama_mmap::impl { throw std::runtime_error(format("CreateFileMappingA failed: %s", llama_format_win_err(error).c_str())); } - // see the POSIX branch: opting in means the eager pull-in is waste - if (llama_mmap_random_mode_get() != LLAMA_MMAP_RANDOM_OFF) { prefetch = 0; } - addr = MapViewOfFile(hMapping, FILE_MAP_READ, 0, 0, 0); DWORD error = GetLastError(); @@ -721,13 +747,55 @@ struct llama_mmap::impl { GGML_UNUSED(last); } - // Windows has no per-mapping "read this randomly" hint. skipping the eager - // PrefetchVirtualMemory in the constructor is what keeps the pages out; there is nothing - // further to say here, and nothing to drop back. - void advise_random(bool drop) { + // Windows has no "read this range randomly" hint. not pulling the range in is what keeps the + // pages out; there is nothing further to say here, and nothing to drop back. + void advise_random_range(size_t offset, size_t len, bool drop) { + GGML_UNUSED(offset); + GGML_UNUSED(len); GGML_UNUSED(drop); } + void prefetch_except(const std::vector> & skip) { +#if _WIN32_WINNT >= 0x602 + BOOL (WINAPI *pPrefetchVirtualMemory) (HANDLE, ULONG_PTR, PWIN32_MEMORY_RANGE_ENTRY, ULONG); + HMODULE hKernel32 = GetModuleHandleW(L"kernel32.dll"); + + pPrefetchVirtualMemory = (decltype(pPrefetchVirtualMemory))(void *) GetProcAddress(hKernel32, "PrefetchVirtualMemory"); + if (!pPrefetchVirtualMemory) { + return; + } + + const size_t page = llama_mmap_page_size(); + + std::vector entries; + size_t pos = 0; + for (const auto & [off, len] : skip) { + const size_t first = off & ~(page - 1); + if (first > pos) { + WIN32_MEMORY_RANGE_ENTRY e; + e.VirtualAddress = (char *) addr + pos; + e.NumberOfBytes = (SIZE_T) (first - pos); + entries.push_back(e); + } + pos = std::max(pos, std::min(size, (off + len + page - 1) & ~(page - 1))); + } + if (pos < size) { + WIN32_MEMORY_RANGE_ENTRY e; + e.VirtualAddress = (char *) addr + pos; + e.NumberOfBytes = (SIZE_T) (size - pos); + entries.push_back(e); + } + + if (!entries.empty() && !pPrefetchVirtualMemory(GetCurrentProcess(), (ULONG_PTR) entries.size(), entries.data(), 0)) { + LLAMA_LOG_WARN("warning: PrefetchVirtualMemory failed: %s\n", + llama_format_win_err(GetLastError()).c_str()); + } +#else + GGML_UNUSED(skip); + LLAMA_LOG_DEBUG("skipping PrefetchVirtualMemory because _WIN32_WINNT < 0x602\n"); +#endif + } + // PrefetchVirtualMemory takes the whole set of ranges in one call, which is exactly the // batching this wants: the reads are issued together instead of one fault at a time. void prefetch_rows(const void * base, size_t stride, size_t row_size, @@ -795,12 +863,20 @@ struct llama_mmap::impl { throw std::runtime_error("mmap not supported"); } - void advise_random(bool drop) { + void advise_random_range(size_t offset, size_t len, bool drop) { + GGML_UNUSED(offset); + GGML_UNUSED(len); GGML_UNUSED(drop); throw std::runtime_error("mmap not supported"); } + void prefetch_except(const std::vector> & skip) { + GGML_UNUSED(skip); + + throw std::runtime_error("mmap not supported"); + } + void prefetch_rows(const void * base, size_t stride, size_t row_size, const int32_t * rows, size_t n_rows) const { GGML_UNUSED(base); @@ -824,8 +900,7 @@ struct llama_mmap::impl { size_t size; // the fd is kept only to re-advise the file; the mapping owns no reference to it - int fd_advise = -1; - bool random = false; + int fd_advise = -1; }; llama_mmap::llama_mmap(struct llama_file * file, size_t prefetch, bool numa) : pimpl(std::make_unique(file, prefetch, numa)) {} @@ -836,22 +911,18 @@ void * llama_mmap::addr() const { return pimpl->addr; } void llama_mmap::unmap_fragment(size_t first, size_t last) { pimpl->unmap_fragment(first, last); } -void llama_mmap::advise_random(bool drop) { - pimpl->advise_random(drop); - // set here rather than per platform: the flag means "the caller opted this mapping in", and - // the batched prefetch is worth doing even where the advice itself is a no-op - pimpl->random = true; +void llama_mmap::advise_random_range(size_t offset, size_t len, bool drop) { + pimpl->advise_random_range(offset, len, drop); } -bool llama_mmap::is_random() const { return pimpl->random; } +void llama_mmap::prefetch_except(const std::vector> & skip) { + pimpl->prefetch_except(skip); +} bool llama_mmap::contains(const void * ptr, size_t len) const { return pimpl->contains(ptr, len); } void llama_mmap::prefetch_rows(const void * base, size_t stride, size_t row_size, const int32_t * rows, size_t n_rows) const { - if (!pimpl->random) { - return; - } pimpl->prefetch_rows(base, stride, row_size, rows, n_rows); } diff --git a/src/llama-mmap.h b/src/llama-mmap.h index ab4ef60408..07a1f6f685 100644 --- a/src/llama-mmap.h +++ b/src/llama-mmap.h @@ -3,6 +3,7 @@ #include #include #include +#include #include struct llama_file; @@ -50,19 +51,21 @@ struct llama_mmap { void unmap_fragment(size_t first, size_t last); - // opt-in, see llama_mmap_random_mode(). marks the whole mapping as randomly accessed, which - // is only correct once loading is done: until then the loader streams the file sequentially. - void advise_random(bool drop); + // opt-in, see llama_mmap_random_mode(). marks one byte range as randomly accessed, which is + // only correct once loading is done: until then the loader streams the file sequentially. + // offsets are into the file, which is also the offset into the mapping - the whole file is + // always mapped from zero. the range is rounded out to whole pages, since madvise needs that. + void advise_random_range(size_t offset, size_t len, bool drop); - // true once advise_random() has been applied. everything that keys off "this mapping is read - // randomly" tests this, so nothing changes for a mapping the user did not opt in for. - bool is_random() const; + // eager pull-in for everything outside the given ranges, used in place of the constructor's + // whole-file one when part of the file must not be read ahead. ranges must be sorted. + void prefetch_except(const std::vector> & skip); // true if [ptr, ptr + len) lies inside this mapping bool contains(const void * ptr, size_t len) const; // ask the kernel to start reading the given rows. issued as one batch so the faults overlap - // instead of serializing; a no-op unless the mapping is marked random. + // instead of serializing. void prefetch_rows(const void * base, size_t stride, size_t row_size, const int32_t * rows, size_t n_rows) const; @@ -78,7 +81,7 @@ private: // host-resident tensors are read sequentially, so this cannot be a default. enum llama_mmap_random_mode { LLAMA_MMAP_RANDOM_OFF = 0, // upstream behaviour - LLAMA_MMAP_RANDOM_ON = 1, // skip MAP_POPULATE/WILLNEED, advise random after load + LLAMA_MMAP_RANDOM_ON = 1, // advise the gather tables random after load, do not pull them in LLAMA_MMAP_RANDOM_DROP = 2, // additionally drop what the load pulled in }; diff --git a/src/llama-model-loader.cpp b/src/llama-model-loader.cpp index 5071556a1f..0654895283 100644 --- a/src/llama-model-loader.cpp +++ b/src/llama-model-loader.cpp @@ -1354,7 +1354,8 @@ void llama_model_loader::init_mappings(bool prefetch, llama_mlocks * mlock_mmaps if (use_mmap) { mappings.reserve(files.size()); mmaps_used.reserve(files.size()); - for (const auto & file : files) { + for (size_t i = 0; i < files.size(); ++i) { + const auto & file = files[i]; bool is_numa = false; auto * dev = ggml_backend_dev_by_type(GGML_BACKEND_DEVICE_TYPE_CPU); @@ -1366,7 +1367,17 @@ void llama_model_loader::init_mappings(bool prefetch, llama_mlocks * mlock_mmaps } } - std::unique_ptr mapping = std::make_unique(file.get(), prefetch ? -1 : 0, is_numa); + const auto no_prefetch = mmap_no_prefetch.find((uint16_t) i); + + // the eager pull-in would read a gather table in full to populate pages the gathers + // hit a few percent of. skip it for this file and ask for everything else instead, + // so the tensors that really are streamed once keep the readahead they had. + const bool split_prefetch = prefetch && !is_numa && no_prefetch != mmap_no_prefetch.end(); + + std::unique_ptr mapping = std::make_unique(file.get(), prefetch && !split_prefetch ? -1 : 0, is_numa); + if (split_prefetch) { + mapping->prefetch_except(no_prefetch->second); + } mmaps_used.emplace_back(mapping->size(), 0); if (mlock_mmaps) { std::unique_ptr mlock_mmap(new llama_mlock()); diff --git a/src/llama-model-loader.h b/src/llama-model-loader.h index e9fe3592d4..96c9d338b1 100644 --- a/src/llama-model-loader.h +++ b/src/llama-model-loader.h @@ -88,6 +88,10 @@ struct llama_model_loader { llama_mmaps mappings; + // byte ranges, per source file, that init_mappings() must not pull in eagerly: gather tables + // the model reads a few percent of. set under LLAMA_MMAP_RANDOM only, sorted by offset. + std::map>> mmap_no_prefetch; + std::map weights_map; std::unordered_map kv_overrides; const llama_model_tensor_buft_override * tensor_buft_overrides; diff --git a/src/llama-model.cpp b/src/llama-model.cpp index 925e2aca61..b4a26830b9 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -1149,9 +1149,16 @@ struct llama_model::impl { // model memory mapped files llama_mmaps mappings; - // set once loading is done, if any mapping was advised random. lets prefetch_rows() bail out - // without walking the mappings, which is the only cost the feature has when it is off. - bool mappings_random = false; + // gather tables that really came out of a mapping, resolved from gather_tables() during load. + // empty unless the user opted in, which is the only cost the feature has when it is off. + struct gather_range { + const ggml_tensor * tensor; + uint16_t idx; // source file, and so the mapping + size_t offs; // byte offset into that file + size_t len; + }; + + std::vector gather_ranges; // objects representing data potentially being locked in memory llama_mlocks mlock_bufs; @@ -1683,6 +1690,22 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) { } } + // kept local until the mappings exist: pimpl->gather_ranges must only ever hold ranges that + // were checked against a live mapping, since everything downstream indexes one + std::vector nominated; + if (llama_mmap_random_mode_get() != LLAMA_MMAP_RANDOM_OFF) { + for (const ggml_tensor * t : gather_tables()) { + const auto * w = t ? ml.get_weight(ggml_get_name(t)) : nullptr; + if (w) { + nominated.push_back({ t, w->idx, w->offs, ggml_nbytes(w->tensor) }); + ml.mmap_no_prefetch[w->idx].emplace_back(w->offs, ggml_nbytes(w->tensor)); + } + } + for (auto & [_, ranges] : ml.mmap_no_prefetch) { + std::sort(ranges.begin(), ranges.end()); + } + } + ml.init_mappings(true, use_mlock ? &pimpl->mlock_mmaps : nullptr); pimpl->mappings.reserve(ml.mappings.size()); @@ -1817,17 +1840,23 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) { pimpl->mappings.emplace_back(std::move(mapping)); } - // only now that every tensor has been read is it safe to say the file is read randomly: + // only now that every tensor has been read is it safe to say a range is read randomly: // the load itself is a sequential pass and wants the readahead it has been getting. const llama_mmap_random_mode random_mode = llama_mmap_random_mode_get(); - if (random_mode != LLAMA_MMAP_RANDOM_OFF) { - for (auto & mapping : pimpl->mappings) { - mapping->advise_random(random_mode == LLAMA_MMAP_RANDOM_DROP); - } - pimpl->mappings_random = !pimpl->mappings.empty(); - LLAMA_LOG_INFO("%s: LLAMA_MMAP_RANDOM: advised %zu mapping(s) for random access%s\n", - __func__, pimpl->mappings.size(), + // a nominated tensor that did not end up served from its mapping was offloaded or copied + // into a buffer, and nothing will gather out of the file. drop it rather than advise it + for (const auto & r : nominated) { + if (r.idx < pimpl->mappings.size() && pimpl->mappings[r.idx]->contains(r.tensor->data, r.len)) { + pimpl->gather_ranges.push_back(r); + } + } + + for (const auto & r : pimpl->gather_ranges) { + pimpl->mappings[r.idx]->advise_random_range(r.offs, r.len, random_mode == LLAMA_MMAP_RANDOM_DROP); + + LLAMA_LOG_INFO("%s: LLAMA_MMAP_RANDOM: %s advised for random access, %.2f MiB%s\n", + __func__, ggml_get_name(r.tensor), r.len / 1024.0 / 1024.0, random_mode == LLAMA_MMAP_RANDOM_DROP ? ", dropped cached pages" : ""); } } @@ -1836,19 +1865,18 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) { } void llama_model::prefetch_rows(const struct ggml_tensor * t, const int32_t * rows, size_t n_rows) const { - if (!pimpl->mappings_random || t == nullptr || t->data == nullptr || n_rows == 0) { + if (pimpl->gather_ranges.empty() || t == nullptr || t->data == nullptr || n_rows == 0) { return; } if (!llama_mmap_random_prefetch_enabled()) { return; } - // rows are addressed off the tensor, so the whole tensor has to sit in the mapping we find - const size_t nbytes = ggml_nbytes(t); - - for (const auto & mapping : pimpl->mappings) { - if (mapping->is_random() && mapping->contains(t->data, nbytes)) { - mapping->prefetch_rows(t->data, t->nb[1], ggml_row_size(t->type, t->ne[0]), rows, n_rows); + // keyed off the tensor, not off its mapping: the readahead must land where the advice did, + // and the mapping now holds ranges that still want the kernel's own readahead + for (const auto & r : pimpl->gather_ranges) { + if (r.tensor == t) { + pimpl->mappings[r.idx]->prefetch_rows(t->data, t->nb[1], ggml_row_size(t->type, t->ne[0]), rows, n_rows); return; } } diff --git a/src/llama-model.h b/src/llama-model.h index cf7c75a476..e101de2994 100644 --- a/src/llama-model.h +++ b/src/llama-model.h @@ -737,11 +737,19 @@ struct llama_model { // ask the kernel to start reading the rows a gather is about to take out of a host-mapped // tensor, so the faults overlap instead of serializing one NVMe latency at a time. // - // does nothing unless the tensor lives in a mapping that was advised random, which only - // happens under LLAMA_MMAP_RANDOM. off, and for anything not read out of a mapping (offloaded - // tensors, --load-mode none, non-POSIX hosts), this is one bool test. + // does nothing unless the tensor was nominated by gather_tables() and really is read out of + // a mapping. off, and for anything else (offloaded tensors, --load-mode none, non-POSIX + // hosts), this is one empty-vector test. void prefetch_rows(const struct ggml_tensor * t, const int32_t * rows, size_t n_rows) const; + // tensors that stay host-resident and are read by sparse row gathers rather than streamed + // once. under LLAMA_MMAP_RANDOM these get the random-access advice and the batched readahead + // of prefetch_rows(); every other tensor keeps the loader's sequential behaviour. + // + // nominated by the model, not guessed from size: a big host-resident tensor read in full, + // such as token_embd on a CPU-only run, wants the readahead this takes away. + virtual std::vector gather_tables() const { return {}; } + float get_rope_freq_base (const llama_cparams & cparams, int il) const; float get_rope_freq_scale(const llama_cparams & cparams, int il) const; diff --git a/src/models/models.h b/src/models/models.h index dacca61600..dc8a279c12 100644 --- a/src/models/models.h +++ b/src/models/models.h @@ -2281,6 +2281,14 @@ struct llama_model_qwen4exp : public llama_model_base { void load_arch_hparams(llama_model_loader & ml) override; void load_arch_tensors(llama_model_loader & ml) override; + // the PLE n-gram table is far too big to offload and is read by 16 tiny gathers per token + std::vector gather_tables() const override { + if (per_layer_tok_embd == nullptr) { + return {}; + } + return { per_layer_tok_embd }; + } + struct graph : public llm_build_delta_net_base { graph(const llama_model & model, const llm_graph_params & params); private: From b8bdf73bb9baf044caadd33be2a51be70156ec57 Mon Sep 17 00:00:00 2001 From: Daniel Han Date: Thu, 27 Aug 2026 09:07:02 +0000 Subject: [PATCH 9/9] llama: fold the random-access prefetch into its own feature flag LLAMA_MMAP_RANDOM_PREFETCH existed to measure the two halves of the feature apart, and the measurement is done: on a cold cache over the same wikitext run, MADV_RANDOM without the batched readahead takes 94.4 s against 36.7 s for an untouched mapping, while the pair together take 34.1 s. Suppressing the kernel's readahead only pays if we replace it, so the split let a user select a 2.6x regression through a documented switch. Keep the accessor, since the call site reads better than a mode comparison, but derive it from the mode alone. --- src/llama-mmap.cpp | 9 +-------- src/llama-mmap.h | 4 +++- 2 files changed, 4 insertions(+), 9 deletions(-) diff --git a/src/llama-mmap.cpp b/src/llama-mmap.cpp index a615eb6ef9..31dc0aa442 100644 --- a/src/llama-mmap.cpp +++ b/src/llama-mmap.cpp @@ -457,14 +457,7 @@ llama_mmap_random_mode llama_mmap_random_mode_get() { } bool llama_mmap_random_prefetch_enabled() { - // on with the feature, so the batched readahead that pays for the random hints cannot be - // left off by accident. separate only so the two halves can be measured apart. - static const bool enabled = []() { - const char * env = getenv("LLAMA_MMAP_RANDOM_PREFETCH"); - return env == nullptr ? true : strcmp(env, "0") != 0; - }(); - - return llama_mmap_random_mode_get() != LLAMA_MMAP_RANDOM_OFF && enabled; + return llama_mmap_random_mode_get() != LLAMA_MMAP_RANDOM_OFF; } static size_t llama_mmap_page_size() { diff --git a/src/llama-mmap.h b/src/llama-mmap.h index 07a1f6f685..b0963dd391 100644 --- a/src/llama-mmap.h +++ b/src/llama-mmap.h @@ -87,7 +87,9 @@ enum llama_mmap_random_mode { llama_mmap_random_mode llama_mmap_random_mode_get(); -// whether batched readahead ahead of a sparse gather is enabled (LLAMA_MMAP_RANDOM_PREFETCH) +// batched readahead ahead of a sparse gather. not separately switchable: MADV_RANDOM suppresses +// the kernel's own readahead, so without this the gather takes a synchronous fault per row and +// runs 2.6x slower than leaving the mapping alone bool llama_mmap_random_prefetch_enabled(); struct llama_mlock {