mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-02 19:07:25 -05:00
remove has_state member, check via ctx_bufs being non-empty
This commit is contained in:
@@ -80,8 +80,6 @@ llama_memory_recurrent::llama_memory_recurrent(
|
||||
continue;
|
||||
}
|
||||
|
||||
has_state = true;
|
||||
|
||||
const char * dev_name = "CPU";
|
||||
|
||||
ggml_backend_buffer_type_t buft = ggml_backend_cpu_buffer_type();
|
||||
@@ -194,8 +192,8 @@ bool llama_memory_recurrent::seq_rm(llama_seq_id seq_id, llama_pos p0, llama_pos
|
||||
|
||||
// partial rollback via per-token snapshot index (bounded by n_rs_seq)
|
||||
if (0 < p0 && p0 <= cell.pos && p1 > cell.pos) {
|
||||
// no layers, so no state to restore, only the position moves back
|
||||
if (!has_state) {
|
||||
// the filter kept no layer (e.g. an MTP draft context), so only the position moves back
|
||||
if (ctxs_bufs.empty()) {
|
||||
cell.pos = p0 - 1;
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -73,9 +73,6 @@ public:
|
||||
// number of recurrent-state snapshots per seq for rollback; tensors are widened to (1 + n_rs_seq) groups
|
||||
uint32_t n_rs_seq = 0;
|
||||
|
||||
// false when the layer filter kept no layer, e.g. an MTP draft context: any rollback is then free
|
||||
bool has_state = false;
|
||||
|
||||
// per-seq rollback index
|
||||
std::vector<uint32_t> rs_idx;
|
||||
|
||||
|
||||
Reference in New Issue
Block a user