From b556d4411c15450c49c40c13c9ee3afe2f4c022a Mon Sep 17 00:00:00 2001 From: bssrdf Date: Sun, 25 Feb 2024 22:55:26 -0500 Subject: [PATCH] fixed bugs; now failed with cuda error; likely out-of-mem on GPU --- clip.hpp | 142 +++++++++++++++++++++++++++++++++++++------ ggml_extend.hpp | 67 +++++++++++++++++++- model.cpp | 12 +++- pmid.hpp | 38 +++++++++--- stable-diffusion.cpp | 6 +- 5 files changed, 232 insertions(+), 33 deletions(-) diff --git a/clip.hpp b/clip.hpp index b9d22e11..a2d2d612 100644 --- a/clip.hpp +++ b/clip.hpp @@ -525,27 +525,61 @@ protected: public: CLIPLayer(int64_t d_model, int64_t n_head, - int64_t intermediate_size) + int64_t intermediate_size, + bool atten1 = true) : d_model(d_model), n_head(n_head), intermediate_size(intermediate_size) { - blocks["self_attn"] = std::shared_ptr(new MultiheadAttention(d_model, n_head)); + if(atten1) + blocks["self_attn"] = std::shared_ptr(new MultiheadAttention(d_model, n_head)); + else + blocks["self_attn"] = std::shared_ptr(new MultiheadAttention2(d_model, n_head)); blocks["layer_norm1"] = std::shared_ptr(new LayerNorm(d_model)); blocks["layer_norm2"] = std::shared_ptr(new LayerNorm(d_model)); blocks["mlp"] = std::shared_ptr(new CLIPMLP(d_model, intermediate_size)); } - struct ggml_tensor* forward(struct ggml_context* ctx, struct ggml_tensor* x, bool mask = true) { + struct ggml_tensor* forward(struct ggml_context* ctx, struct ggml_tensor* x, bool mask = true, bool atten1 = true) { // x: [N, n_token, d_model] - auto self_attn = std::dynamic_pointer_cast(blocks["self_attn"]); - auto layer_norm1 = std::dynamic_pointer_cast(blocks["layer_norm1"]); - auto layer_norm2 = std::dynamic_pointer_cast(blocks["layer_norm2"]); - auto mlp = std::dynamic_pointer_cast(blocks["mlp"]); - - x = ggml_add(ctx, x, self_attn->forward(ctx, layer_norm1->forward(ctx, x), mask)); - x = ggml_add(ctx, x, mlp->forward(ctx, layer_norm2->forward(ctx, x))); - return x; + if(atten1){ + auto self_attn = std::dynamic_pointer_cast(blocks["self_attn"]); + auto layer_norm1 = std::dynamic_pointer_cast(blocks["layer_norm1"]); + auto layer_norm2 = std::dynamic_pointer_cast(blocks["layer_norm2"]); + auto mlp = std::dynamic_pointer_cast(blocks["mlp"]); + + // struct ggml_tensor* r = x; + + struct ggml_tensor* h = layer_norm1->forward(ctx, x); + + h = self_attn->forward(ctx, h, mask); + + + // x = ggml_add(ctx, x, self_attn->forward(ctx, layer_norm1->forward(ctx, x), mask)); + x = ggml_add(ctx, x, h); + + x = ggml_add(ctx, x, mlp->forward(ctx, layer_norm2->forward(ctx, x))); + return x; + } + else{ + auto self_attn = std::dynamic_pointer_cast(blocks["self_attn"]); + auto layer_norm1 = std::dynamic_pointer_cast(blocks["layer_norm1"]); + auto layer_norm2 = std::dynamic_pointer_cast(blocks["layer_norm2"]); + auto mlp = std::dynamic_pointer_cast(blocks["mlp"]); + + // struct ggml_tensor* r = x; + + struct ggml_tensor* h = layer_norm1->forward(ctx, x); + + h = self_attn->forward(ctx, h, mask); + + + // x = ggml_add(ctx, x, self_attn->forward(ctx, layer_norm1->forward(ctx, x), mask)); + x = ggml_add(ctx, x, h); + + x = ggml_add(ctx, x, mlp->forward(ctx, layer_norm2->forward(ctx, x))); + return x; + } } }; @@ -557,15 +591,16 @@ public: CLIPEncoder(int64_t n_layer, int64_t d_model, int64_t n_head, - int64_t intermediate_size) + int64_t intermediate_size, + bool atten1 = true) : n_layer(n_layer) { for (int i = 0; i < n_layer; i++) { - std::string name = "layers." + std::to_string(i); - blocks[name] = std::shared_ptr(new CLIPLayer(d_model, n_head, intermediate_size)); + std::string name = "layers." + std::to_string(i); + blocks[name] = std::shared_ptr(new CLIPLayer(d_model, n_head, intermediate_size, atten1)); } } - struct ggml_tensor* forward(struct ggml_context* ctx, struct ggml_tensor* x, int clip_skip = -1, bool mask = true) { + struct ggml_tensor* forward(struct ggml_context* ctx, struct ggml_tensor* x, int clip_skip = -1, bool mask = true, bool atten1 = true) { // x: [N, n_token, d_model] int layer_idx = n_layer - 1; LOG_DEBUG("clip_skip %d", clip_skip); @@ -579,8 +614,10 @@ public: break; } std::string name = "layers." + std::to_string(i); + // printf(" about to do %s\n", name.c_str()); auto layer = std::dynamic_pointer_cast(blocks[name]); - x = layer->forward(ctx, x); // [N, n_token, d_model] + x = layer->forward(ctx, x, mask, atten1); // [N, n_token, d_model] + // print_ggml_tensor(x, true, ("layer "+std::to_string(i)).c_str()); // LOG_DEBUG("layer %d", i); } return x; @@ -779,7 +816,7 @@ public: class CLIPVisionModel : public GGMLBlock { protected: void init_params(struct ggml_context* ctx, ggml_type wtype) { - params["visual_projection"] = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, projection_dim, hidden_size); + params["visual_projection.weight"] = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, projection_dim, hidden_size); } public: @@ -837,10 +874,77 @@ public: } return pooled; // [N, projection_dim] } +}; + +class CLIPVisionModel2 : public GGMLBlock { +protected: + void init_params(struct ggml_context* ctx, ggml_type wtype) { + params["visual_projection.weight"] = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, hidden_size, projection_dim); + } + +public: + // network hparams + int32_t num_channels = 3; + int32_t patch_size = 14; + int32_t image_size = 224; + int32_t num_positions = 257; // (image_size / patch_size)^2 + 1 + int32_t hidden_size = 1024; + int32_t intermediate_size = 4096; + int32_t n_head = 16; + int32_t n_layer = 24; + int32_t projection_dim = 768; + +public: + CLIPVisionModel2(CLIPVersion version = OPENAI_CLIP_VIT_L_14) { + if (version == OPEN_CLIP_VIT_H_14) { + hidden_size = 1280; + intermediate_size = 5120; + n_head = 16; + n_layer = 32; + projection_dim = 1024; + } else if (version == OPEN_CLIP_VIT_BIGG_14) { + hidden_size = 1664; + intermediate_size = 8192; + n_head = 16; + n_layer = 48; + } + + blocks["embeddings"] = std::shared_ptr(new CLIPVisionEmbeddings(hidden_size, num_channels, patch_size, image_size)); + blocks["pre_layernorm"] = std::shared_ptr(new LayerNorm(hidden_size)); + blocks["encoder"] = std::shared_ptr(new CLIPEncoder(n_layer, hidden_size, n_head, intermediate_size, false)); + blocks["post_layernorm"] = std::shared_ptr(new LayerNorm(hidden_size)); + } + + struct ggml_tensor* forward(struct ggml_context* ctx, struct ggml_tensor* pixel_values) { + // pixel_values: [N, num_channels, image_size, image_size] + // return: // [N, projection_dim] + auto embeddings = std::dynamic_pointer_cast(blocks["embeddings"]); + auto pre_layernorm = std::dynamic_pointer_cast(blocks["pre_layernorm"]); + auto encoder = std::dynamic_pointer_cast(blocks["encoder"]); + auto post_layernorm = std::dynamic_pointer_cast(blocks["post_layernorm"]); + + auto x = embeddings->forward(ctx, pixel_values); // [N, num_positions, embed_dim] + print_ggml_tensor(x, true, "embedding"); + x = pre_layernorm->forward(ctx, x); + print_ggml_tensor(x, true, "pre_layernorm"); + x = encoder->forward(ctx, x, -1, false, false); + print_ggml_tensor(x, true, "encoder"); + x = post_layernorm->forward(ctx, x); // [N, n_token, hidden_size] + print_ggml_tensor(x, true, "post_layernorm"); + + GGML_ASSERT(x->ne[3] == 1); + // int64_t max_token_idx = 0; + // ggml_tensor* pooled = ggml_view_1d(ctx, x, x->ne[0], x->nb[1] * max_token_idx); // assert N == 1 + x = ggml_cont(ctx, ggml_permute(ctx, x, 0, 2, 1, 3)); + // ggml_view_2d(ctx, w, w->ne[0], w->ne[1] / 2, w->nb[1], 0); + ggml_tensor* pooled = ggml_view_2d(ctx, x, x->ne[0], x->ne[1], x->nb[1], 0); + print_ggml_tensor(pooled, true, "pooled"); + return pooled; // [N, projection_dim] + } struct ggml_tensor* visual_project(struct ggml_context* ctx, struct ggml_tensor* input){ - auto visual_projection = params["visual_projection"]; - auto pooled = ggml_mul_mat(ctx, ggml_cont(ctx, ggml_transpose(ctx, visual_projection)), input); + auto visual_projection = params["visual_projection.weight"]; + auto pooled = ggml_mul_mat(ctx, visual_projection, input); return pooled; } diff --git a/ggml_extend.hpp b/ggml_extend.hpp index 49af792c..2761a70c 100644 --- a/ggml_extend.hpp +++ b/ggml_extend.hpp @@ -1037,13 +1037,16 @@ public: } for (auto& pair : blocks) { auto& block = pair.second; - + // if(starts_with(prefix, "pmid")) + // printf("block pair.first: %s, %s \n", prefix.c_str(), pair.first.c_str()); + block->get_param_tensors(tensors, prefix + pair.first); } for (auto& pair : params) { struct ggml_tensor* param = pair.second; - + // if(starts_with(prefix, "pmid")) + // printf("params pair.first: %s, %s \n", prefix.c_str(), pair.first.c_str()); tensors[prefix + pair.first] = pair.second; } } @@ -1313,4 +1316,64 @@ public: } }; + +class MultiheadAttention2 : public GGMLBlock { +protected: + int64_t embed_dim; + int64_t n_head; + bool bias; + bool mask; + +public: + MultiheadAttention2(int64_t embed_dim, + int64_t n_head, + bool bias = true) + : embed_dim(embed_dim), + n_head(n_head), + bias(bias) { + // printf("MultiheadAttention2: %d, %d, %d\n", embed_dim, n_head, bias?1:0); + blocks["q_proj"] = std::shared_ptr(new Linear(embed_dim, embed_dim, bias)); + blocks["k_proj"] = std::shared_ptr(new Linear(embed_dim, embed_dim, bias)); + blocks["v_proj"] = std::shared_ptr(new Linear(embed_dim, embed_dim, bias)); + blocks["out_proj"] = std::shared_ptr(new Linear(embed_dim, embed_dim, bias)); + } + + // x: [N, n_token, embed_dim] + struct ggml_tensor* forward(struct ggml_context* ctx, struct ggml_tensor* x, bool mask = false) { + auto q_proj = std::dynamic_pointer_cast(blocks["q_proj"]); + auto k_proj = std::dynamic_pointer_cast(blocks["k_proj"]); + auto v_proj = std::dynamic_pointer_cast(blocks["v_proj"]); + auto out_proj = std::dynamic_pointer_cast(blocks["out_proj"]); + + int64_t N = x->ne[2]; + int64_t n_token = x->ne[1]; + int64_t d_head = embed_dim / n_head; + + struct ggml_tensor* q = q_proj->forward(ctx, x); + q = ggml_reshape_4d(ctx, q, d_head, n_head, n_token, N); // [N, n_token, n_head, d_head] + q = ggml_cont(ctx, ggml_permute(ctx, q, 0, 2, 1, 3)); // [N, n_head, n_token, d_head] + q = ggml_reshape_3d(ctx, q, d_head, n_token, n_head * N); // [N * n_head, n_token, d_head] + + struct ggml_tensor* k = k_proj->forward(ctx, x); + k = ggml_reshape_4d(ctx, k, d_head, n_head, n_token, N); // [N, n_token, n_head, d_head] + k = ggml_cont(ctx, ggml_permute(ctx, k, 0, 2, 1, 3)); // [N, n_head, n_token, d_head] + // k = ggml_reshape_3d(ctx, k, d_head, n_token, n_head); // [N * n_head, n_token, d_head] !!BUG!! + k = ggml_reshape_3d(ctx, k, d_head, n_token, n_head * N); // [N * n_head, n_token, d_head] + + struct ggml_tensor* v = v_proj->forward(ctx, x); + v = ggml_reshape_4d(ctx, v, d_head, n_head, n_token, N); // [N, n_token, n_head, d_head] + v = ggml_cont(ctx, ggml_permute(ctx, v, 1, 2, 0, 3)); // [N, n_head, d_head, n_token] + v = ggml_reshape_3d(ctx, v, n_token, d_head, n_head * N); // [N * n_head, d_head, n_token] + + struct ggml_tensor* kqv = ggml_nn_attention(ctx, q, k, v, mask); // [N * n_head, n_token, d_head] + + kqv = ggml_reshape_4d(ctx, kqv, d_head, n_token, n_head, N); + kqv = ggml_cont(ctx, ggml_permute(ctx, kqv, 0, 2, 1, 3)); // [N, n_token, n_head, d_head] + + x = ggml_reshape_3d(ctx, kqv, d_head * n_head, n_token, N); // [N * n_token, d_head * n_head] + x = out_proj->forward(ctx, x); + return x; + } +}; + #endif // __GGML_EXTEND__HPP__ \ No newline at end of file diff --git a/model.cpp b/model.cpp index c3e68780..12673de6 100644 --- a/model.cpp +++ b/model.cpp @@ -828,7 +828,17 @@ bool ModelLoader::init_from_safetensors_file(const std::string& file_path, const return false; } } - + // fix photomaker model file to remove the following name fix + size_t pos = name.find("layrnorm"); + if(pos != std::string::npos){ + name.replace(pos, strlen("layrnorm"), "layernorm"); + } + if(starts_with(prefix, "pmid")){ + pos = name.find("visual_projection.w"); + if(pos != std::string::npos){ + name.replace(0, 0, "vision_model."); + } + } TensorStorage tensor_storage(prefix + name, type, ne, n_dims, file_index, ST_HEADER_SIZE_LEN + header_size_ + begin); tensor_storage.reverse_ne(); diff --git a/pmid.hpp b/pmid.hpp index d6120db7..8224d50c 100644 --- a/pmid.hpp +++ b/pmid.hpp @@ -34,7 +34,7 @@ public: // blocks["fc2"] = std::shared_ptr(new Linear(intermediate_size, d_model)); blocks["fc1"] = std::shared_ptr(new Linear(in_dim, hidden_dim, true)); blocks["fc2"] = std::shared_ptr(new Linear(hidden_dim, out_dim, true)); - blocks["layer_norm"] = std::shared_ptr(new LayerNorm(in_dim)); + blocks["layernorm"] = std::shared_ptr(new LayerNorm(in_dim)); } @@ -97,7 +97,7 @@ public: auto fc1 = std::dynamic_pointer_cast(blocks["fc1"]); auto fc2 = std::dynamic_pointer_cast(blocks["fc2"]); - auto layer_norm = std::dynamic_pointer_cast(blocks["layer_norm"]); + auto layer_norm = std::dynamic_pointer_cast(blocks["layernorm"]); struct ggml_tensor* r = x; // x = ggml_nn_layer_norm(ctx, x, ln_w, ln_b); @@ -177,6 +177,12 @@ public: // return mem_size; // } + // void get_param_tensors(std::map& tensors, const std::string prefix) { + // vision_model.get_param_tensors(tensors, prefix+".vision_model"); + // fuse_module.get_param_tensors(tensors, prefix+".fuse_module"); + // visual_projection_2.get_param_tensors(tensors, prefix); + // } + struct ggml_tensor* fuse_fn(struct ggml_context* ctx, struct ggml_tensor* prompt_embeds, struct ggml_tensor* id_embeds) { @@ -192,6 +198,7 @@ public: auto stacked_id_embeds = ggml_concat(ctx, prompt_embeds0, id_embeds0); stacked_id_embeds = ggml_cont(ctx, ggml_permute(ctx, stacked_id_embeds, 1, 2, 0, 3)); + // stacked_id_embeds = mlp1.forward(ctx, stacked_id_embeds); // stacked_id_embeds = ggml_add(ctx, stacked_id_embeds, prompt_embeds); // stacked_id_embeds = mlp2.forward(ctx, stacked_id_embeds); @@ -219,7 +226,6 @@ public: struct ggml_tensor * valid_id_embeds = id_embeds; // # slice out the image token embeddings struct ggml_tensor * image_token_embeds = ggml_get_rows(ctx, prompt_embeds, class_tokens_mask_pos); - struct ggml_tensor *stacked_id_embeds = fuse_fn(ctx, image_token_embeds, valid_id_embeds); stacked_id_embeds = ggml_cont(ctx, ggml_permute(ctx, stacked_id_embeds, 0, 2, 1, 3)); @@ -254,11 +260,18 @@ public: blocks["visual_projection_2"] = std::shared_ptr(new Linear(hidden_size, projection_dim, false)); } + // void init_params(struct ggml_context* ctx, ggml_type wtype) { + // params["visual_projection_2.weight"] = ggml_new_tensor_2d(ctx, wtype, hidden_size, projection_dim); + + // } + struct ggml_tensor* forward(struct ggml_context* ctx, struct ggml_tensor* pixel_values){ auto visual_projection_2 = std::dynamic_pointer_cast(blocks["visual_projection_2"]); + // auto visual_projection_2 = params["visual_projection_2.weight"]; auto x = visual_projection_2->forward(ctx, pixel_values); // [N, projection_dim] + // auto x = ggml_mul_mat(ctx, visual_projection_2, pixel_values); return x; // [N, projection_dim] } @@ -270,7 +283,7 @@ struct PhotoMakerIDEncoder : public GGMLModule { public: SDVersion version = VERSION_XL; - CLIPVisionModel vision_model; + CLIPVisionModel2 vision_model; FuseModule fuse_module; // struct ggml_tensor* visual_projection_2; VisualProjection visual_projection_2; @@ -286,8 +299,11 @@ public: style_strength(sty){ // vision_model = CLIPVisionModel(); vision_model.init(params_ctx, wtype); + fuse_module.init(params_ctx, wtype); + visual_projection_2.init(params_ctx, wtype); } + std::string get_desc() { return "pmid"; } @@ -342,8 +358,8 @@ public: void get_param_tensors(std::map& tensors, const std::string prefix) { - vision_model.get_param_tensors(tensors, prefix); - fuse_module.get_param_tensors(tensors, prefix); + vision_model.get_param_tensors(tensors, prefix+".vision_model"); + fuse_module.get_param_tensors(tensors, prefix+".fuse_module"); visual_projection_2.get_param_tensors(tensors, prefix); } @@ -365,14 +381,15 @@ public: // class_embedding_temp, // positions // ); // [batch_size, seq_length, hidden_size] - struct ggml_tensor *shared_id_embeds = vision_model.forward(ctx, - id_pixel_values, false); // [batch_size, seq_length, hidden_size] - + struct ggml_tensor *shared_id_embeds = vision_model.forward(ctx, id_pixel_values); // [batch_size, seq_length, hidden_size] + print_ggml_tensor(shared_id_embeds, true, "shared_id_embeds"); struct ggml_tensor *id_embeds = vision_model.visual_project(ctx, shared_id_embeds); // [batch_size, seq_length, proj_dim(768)] + print_ggml_tensor(id_embeds, true, "id_embeds"); // struct ggml_tensor *id_embeds_2 = ggml_mul_mat(ctx, visual_projection_2, shared_id_embeds); // [batch_size, seq_length, 1280] struct ggml_tensor *id_embeds_2 = visual_projection_2.forward(ctx, shared_id_embeds); // [batch_size, seq_length, 1280] + print_ggml_tensor(id_embeds_2, true, "id_embeds_2"); @@ -382,11 +399,14 @@ public: id_embeds = ggml_concat(ctx, id_embeds, id_embeds_2); // [batch_size, seq_length, 1, 2048] check whether concat at dim 2 is right id_embeds = ggml_cont(ctx, ggml_permute(ctx, id_embeds, 1, 2, 0, 3)); + print_ggml_tensor(id_embeds, true, "id_embeds_after_cont+perm"); + struct ggml_tensor * updated_prompt_embeds = fuse_module.forward(ctx, prompt_embeds, id_embeds, class_tokens_mask, class_tokens_mask_pos, left, right); + print_ggml_tensor(updated_prompt_embeds, true, "updated_prompt_embeds"); return updated_prompt_embeds; diff --git a/stable-diffusion.cpp b/stable-diffusion.cpp index 0d249909..4dafe8d9 100644 --- a/stable-diffusion.cpp +++ b/stable-diffusion.cpp @@ -278,7 +278,7 @@ public: } // LOG_INFO("pmid param memory buffer size = %.2fMB ", // pmid_model->params_buffer_size / 1024.0 / 1024.0); - pmid_model->get_param_tensors(tensors, "pmid."); + pmid_model->get_param_tensors(tensors, "pmid"); } // if(stacked_id){ // pmid_model.init_params(GGML_TYPE_F32); @@ -1643,6 +1643,7 @@ sd_image_t* txt2img(sd_ctx_t* sd_ctx, // if (sd_ctx->sd->free_params_immediately) { // sd_ctx->sd->pmid_lora.free_params_buffer(); // } + LOG_INFO("pmid_lora apply completed"); } struct ggml_init_params params; @@ -1693,12 +1694,13 @@ sd_image_t* txt2img(sd_ctx_t* sd_ctx, auto cond_tup = sd_ctx->sd->get_learned_condition_with_trigger(work_ctx, prompt, clip_skip, width, height, num_input_images ); + LOG_INFO("get_learned_condition_with_trigger finished"); prompts_embeds = std::get<0>(cond_tup); pooled_prompts_embeds = std::get<1>(cond_tup); // [adm_in_channels, ] class_tokens_mask = std::get<2>(cond_tup); // prompts_embeds = sd_ctx->sd->id_encoder(work_ctx, init_img, prompts_embeds, class_tokens_mask); - + LOG_INFO("id_encoder"); // if (sd_ctx->sd->free_params_immediately) { // sd_ctx->sd->pmid_model.free_params_buffer(); // }