llama : fix K/V and recurrent state cleanup after failed restores (#27530)

* llama : add discard for deferred state writes

* llama : add tensor zeroing helper for backends without tensor memset

* llama : clear K/V data after failed sequence restore

* llama : clear recurrent state data after failed sequence restore

* llama : simplify discard and restore cleanup

* llama : report error when abnormal cell count is found in state_read_meta

* llama : clear attention state on hybrid restore failure

* tests : cover failed state restore cleanup

* llama : clear MLA state on dsa restore failure

* tests : update test for rebased test suite

* llama : clarify comment in llama_memory_recurrent::state_read
This commit is contained in:
Chipmunk
2026-09-26 10:23:03 +03:00
committed by GitHub
parent a1de614ba3
commit 08618ff8e7
11 changed files with 383 additions and 17 deletions
+12
View File
@@ -1,8 +1,10 @@
#include "llama-impl.h"
#include "ggml-backend.h"
#include "gguf.h"
#include "llama.h"
#include <algorithm>
#include <cinttypes>
#include <climits>
#include <cstdarg>
@@ -66,6 +68,16 @@ void llama_log_callback_default(ggml_log_level level, const char * text, void *
fflush(stderr);
}
void llama_clear_tensor_data(ggml_tensor * t, size_t offset, size_t size) {
static const std::vector<uint8_t> zeros(1024*1024, 0);
// not all backend buffers implement ggml_backend_tensor_memset(), so write zeros instead
// TODO: make this a generic fallback in `ggml_backend_tensor_memset` when `set_tensor` is available
for (size_t ofs = 0; ofs < size; ofs += zeros.size()) {
ggml_backend_tensor_set(t, zeros.data(), offset + ofs, std::min(size - ofs, zeros.size()));
}
}
void replace_all(std::string & s, const std::string & search, const std::string & replace) {
if (search.empty()) {
return;