diff --git a/include/llama.h b/include/llama.h index ce454df527..e808fa9ec1 100644 --- a/include/llama.h +++ b/include/llama.h @@ -482,14 +482,13 @@ extern "C" { LLAMA_API struct llama_model_quantize_params llama_model_quantize_default_params(void); // Initialize the llama + ggml backend - // If numa is true, use NUMA optimizations // Call once at the start of the program LLAMA_API void llama_backend_init(void); // Call once at the end of the program - currently only used for MPI LLAMA_API void llama_backend_free(void); - //optional: + // Optional: enable numa optimizations LLAMA_API void llama_numa_init(enum ggml_numa_strategy numa); // Optional: an auto threadpool gets created in ggml if not passed explicitly diff --git a/src/llama.cpp b/src/llama.cpp index ad8e443882..e97272c889 100644 --- a/src/llama.cpp +++ b/src/llama.cpp @@ -401,6 +401,7 @@ static struct llama_model * llama_model_load_from_file_impl( return nullptr; } } + // TODO: remove ggml_time_init(); if (!params.vocab_only && ggml_backend_reg_count() == 0) { diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 996e7f58f3..01af4ae373 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -316,6 +316,7 @@ llama_build_and_test(test-backend-sampler.cpp LABEL "model") # Test for state restore with fragmented KV cache # Requires a model, uses same args pattern as test-thread-safety +# TODO: run on all dummy models llama_build_and_test(test-state-restore-fragmented.cpp LABEL "model" ARGS -m "${MODEL_DEST}") set_tests_properties(test-state-restore-fragmented PROPERTIES FIXTURES_REQUIRED test-download-model) diff --git a/tests/test-recurrent-state-rollback.cpp b/tests/test-recurrent-state-rollback.cpp index fdac344d89..1944a441cb 100644 --- a/tests/test-recurrent-state-rollback.cpp +++ b/tests/test-recurrent-state-rollback.cpp @@ -495,7 +495,7 @@ int main(int argc, char ** argv) { return 1; } - ggml_backend_load_all(); + llama_backend_init(); common_init_result_ptr llama_init = common_init_from_params(params); llama_model * model = llama_init->model(); diff --git a/tests/test-save-load-state.cpp b/tests/test-save-load-state.cpp index 083ad6a6ca..dee5e17be6 100644 --- a/tests/test-save-load-state.cpp +++ b/tests/test-save-load-state.cpp @@ -913,7 +913,7 @@ int main(int argc, char ** argv) { params.n_predict = 16; } - ggml_backend_load_all(); + llama_backend_init(); if (!models_dir.empty()) { // run the suite over every dummy model in the directory diff --git a/tests/test-state-restore-fragmented.cpp b/tests/test-state-restore-fragmented.cpp index d5548afba1..33ce6f2763 100644 --- a/tests/test-state-restore-fragmented.cpp +++ b/tests/test-state-restore-fragmented.cpp @@ -30,7 +30,7 @@ int main(int argc, char ** argv) { // init - ggml_backend_load_all(); + llama_backend_init(); common_init_result_ptr llama_init = common_init_from_params(params);