From dbe74df6a518f7fae9baebec88292ccfc508b5ad Mon Sep 17 00:00:00 2001 From: bssrdf Date: Thu, 22 Feb 2024 21:52:01 -0500 Subject: [PATCH] removing image normalization step seems making ID fidelity much higher --- ggml_extend.hpp | 17 +++-------------- model.cpp | 6 +++--- stable-diffusion.cpp | 42 ++++++++++++++++++++++++++++++++++++++---- 3 files changed, 44 insertions(+), 21 deletions(-) diff --git a/ggml_extend.hpp b/ggml_extend.hpp index a0fcedef..6f7f4504 100644 --- a/ggml_extend.hpp +++ b/ggml_extend.hpp @@ -706,18 +706,6 @@ struct GGMLModule { for (int i = 0; i < gf->n_leafs; i++) { struct ggml_tensor * t1 = gf->leafs[i]; - if(strcmp(ggml_get_name(t1), "class_tokens_mask_pos_input") == 0) { - imb = t1; - int64_t stride = imb->ne[0]; - int* out_data = new int[ggml_nelements(imb)]; - ggml_backend_tensor_get(imb, out_data, 0, ggml_nbytes(imb)); - printf("["); - for(int i = 0; i < stride; i++){ - printf("%d, ", out_data[i]); - } - printf("]\n"); - delete out_data; - } // if(strcmp(ggml_get_name(t1), "prompt_embeds_input") == 0) { if(strcmp(ggml_get_name(t1), "id_pixel_values_input") == 0) { @@ -796,8 +784,9 @@ struct GGMLModule { delete out_data; } } -//#if 0 -//#endif +#endif +#if 0 + // struct ggml_tensor * imb = NULL; for (int i = 0; i < gf->n_nodes; i++) { struct ggml_cgraph g1v = ggml_graph_view(gf, i, i + 1); diff --git a/model.cpp b/model.cpp index 8b3036cf..6663aa88 100644 --- a/model.cpp +++ b/model.cpp @@ -789,8 +789,8 @@ bool ModelLoader::init_from_safetensors_file(const std::string& file_path, const ne[i] = shape[i].get(); } TensorStorage tensor_storage(prefix + name, type, ne, n_dims, file_index, ST_HEADER_SIZE_LEN + header_size_ + begin); - // if(starts_with(prefix, "pmi")) - // printf(" %s:%s [%zu, %zu, %zu, %zu]\n", (prefix + name).c_str(), ggml_type_name(type), ne[0], ne[1],ne[2], ne[3]); + // if(starts_with(prefix, "scar")) + // printf(" %s:%s [%zu, %zu, %zu, %zu]\n", (prefix + name).c_str(), ggml_type_name(type), ne[0], ne[1],ne[2], ne[3]); tensor_storage.reverse_ne(); size_t tensor_data_size = end - begin; @@ -1127,7 +1127,7 @@ bool ModelLoader::parse_data_pkl(uint8_t* buffer, if (reader.phase == PickleTensorReader::READ_DIMENS) { reader.tensor_storage.reverse_ne(); reader.tensor_storage.file_index = file_index; - // if(strcmp(prefix.c_str(), "pm") == 0) + // if(strcmp(prefix.c_str(), "scarlett") == 0) // printf(" got tensor %s \n ", reader.tensor_storage.name.c_str()); reader.tensor_storage.name = prefix + reader.tensor_storage.name; tensor_storages.push_back(reader.tensor_storage); diff --git a/stable-diffusion.cpp b/stable-diffusion.cpp index efad98c6..1f5a9287 100644 --- a/stable-diffusion.cpp +++ b/stable-diffusion.cpp @@ -270,7 +270,7 @@ public: if(stacked_id){ pmid_model.init_params(GGML_TYPE_F32); - pmid_model.map_by_name(tensors, "pmid."); + pmid_model.map_by_name(tensors, "pmid."); } } @@ -648,6 +648,21 @@ public: pmid_model.alloc_compute_buffer(work_ctx, init_img, prompts_embeds, class_tokens_mask); pmid_model.compute(n_threads, init_img, prompts_embeds, class_tokens_mask, res); pmid_model.free_compute_buffer(); + + // float original_mean = ggml_tensor_mean(res); + // for (int i2 = 0; i2 < res->ne[2]; i2++) { + // for (int i1 = 0; i1 < res->ne[1]; i1++) { + // if (class_tokens_mask[i1]){ + // for (int i0 = 0; i0 < res->ne[0]; i0++) { + // float value = ggml_tensor_get_f32(res, i0, i1, i2); + // value *= 1.1f; + // ggml_tensor_set_f32(res, value, i0, i1, i2); + // } + // } + // } + // } + // float new_mean = ggml_tensor_mean(res); + // ggml_tensor_scale(res, (original_mean / new_mean)); // for(int j = 0; j < cond_stage_model.text_model.max_position_embeddings; j++){ // printf("["); @@ -1530,9 +1545,28 @@ sd_image_t* txt2img(sd_ctx_t* sd_ctx, float std[] = {0.26862954, 0.26130258, 0.27577711}; for(int i = 0; i < num_input_images; i++) { sd_image_t* init_image = input_id_images[i]; - sd_mul_images_to_tensor(init_image->data, init_img, i, mean, std); - // sd_mul_images_to_tensor(init_image->data, init_img, i, NULL, NULL); - } + // sd_mul_images_to_tensor(init_image->data, init_img, i, mean, std); + sd_mul_images_to_tensor(init_image->data, init_img, i, NULL, NULL); + } + + // ModelLoader model_loader; + // std::string img_path("examples/scarlett.safetensors"); + // if (!model_loader.init_from_file(img_path, "scarlett.")){ + // LOG_ERROR("init model loader from file failed: '%s'", img_path.c_str()); + // return NULL; + // } + // ggml_backend_t cpu_backend = ggml_backend_cpu_init(); + + // std::map tensors_need_to_load; + // std::set ignore_tensors; + // tensors_need_to_load["scarlett.img"] = init_img; + // bool success = model_loader.load_tensors(tensors_need_to_load, cpu_backend, ignore_tensors); + // if (!success) { + // LOG_ERROR("load tensors from model loader failed"); + // ggml_free(work_ctx); + // return NULL; + // } + // sd_image_t* cropped_images = (sd_image_t*)calloc(num_input_images, sizeof(sd_image_t)); // if (cropped_images == NULL) { // ggml_free(work_ctx);