Compare commits

...
35 changed files with 1181 additions and 265 deletions
+1
View File
@@ -64,6 +64,7 @@ API and command-line option may change frequently.***
- [HiDream-O1-Image](./docs/hidream_o1_image.md)
- [Ideogram4](./docs/ideogram4.md)
- [LLaDA-Image](./docs/llada_image.md)
- [Ming-Image Design](./docs/ming_image.md)
- [PixArt](./docs/pixart.md)
- [Image Edit Models](./docs/edit.md)
- [FLUX.1-Kontext-dev](./docs/kontext.md)
Binary file not shown.
+2
View File
@@ -40,4 +40,6 @@ RUN printf '#!/bin/sh\nexec /sd.cpp/bin/sd-cli "$@"\n' > /sd-cli && \
printf '#!/bin/sh\nexec /sd.cpp/bin/sd-server "$@"\n' > /sd-server && \
chmod +x /sd-cli /sd-server
WORKDIR /sd.cpp
ENTRYPOINT [ "/sd-cli" ]
+2
View File
@@ -57,4 +57,6 @@ RUN printf '#!/bin/sh\nexec /sd.cpp/bin/sd-cli "$@"\n' > /sd-cli && \
printf '#!/bin/sh\nexec /sd.cpp/bin/sd-server "$@"\n' > /sd-server && \
chmod +x /sd-cli /sd-server
WORKDIR /sd.cpp
ENTRYPOINT [ "/sd-cli" ]
+2
View File
@@ -42,4 +42,6 @@ RUN printf '#!/bin/sh\nexec /sd.cpp/bin/sd-cli "$@"\n' > /sd-cli && \
printf '#!/bin/sh\nexec /sd.cpp/bin/sd-server "$@"\n' > /sd-server && \
chmod +x /sd-cli /sd-server
WORKDIR /sd.cpp
ENTRYPOINT [ "/sd-cli" ]
+2
View File
@@ -29,4 +29,6 @@ FROM intel/oneapi-basekit:${SYCL_VERSION}-devel-ubuntu24.04 AS runtime
COPY --from=build /sd.cpp/build/bin/sd-cli /sd-cli
COPY --from=build /sd.cpp/build/bin/sd-server /sd-server
WORKDIR /sd.cpp
ENTRYPOINT [ "/sd-cli" ]
+2
View File
@@ -41,4 +41,6 @@ RUN printf '#!/bin/sh\nexec /sd.cpp/bin/sd-cli "$@"\n' > /sd-cli && \
printf '#!/bin/sh\nexec /sd.cpp/bin/sd-server "$@"\n' > /sd-server && \
chmod +x /sd-cli /sd-server
WORKDIR /sd.cpp
ENTRYPOINT [ "/sd-cli" ]
+2 -14
View File
@@ -4,20 +4,8 @@
## Download weights
- Download Mage-Flow
- safetensors: https://huggingface.co/microsoft/Mage-Flow/tree/main/transformer
- Download Mage-Flow-Base
- safetensors: https://huggingface.co/microsoft/Mage-Flow-Base/tree/main/transformer
- Download Mage-Flow-Turbo
- safetensors: https://huggingface.co/microsoft/Mage-Flow-Turbo/tree/main/transformer
- Download Mage-Flow-Edit
- safetensors: https://huggingface.co/microsoft/Mage-Flow-Edit/tree/main/transformer
- Download Mage-Flow-Edit-Turbo
- safetensors: https://huggingface.co/microsoft/Mage-Flow-Edit-Turbo/tree/main/transformer
- Download Mage-Flow-Edit-Base
- safetensors: https://huggingface.co/microsoft/Mage-Flow-Edit-Base/tree/main/transformer
- Download Mage-Flow vae
- safetensors: https://huggingface.co/microsoft/Mage-Flow/tree/main/vae
- Download Mage-Flow diffusion from https://huggingface.co/Comfy-Org/Mage-Flow/tree/main/diffusion_models
- Download Mage-Flow vae from https://huggingface.co/Comfy-Org/Mage-Flow/tree/main/vae
- Download Qwen3-VL 4B
- safetensors: https://huggingface.co/Comfy-Org/Krea-2/tree/main/text_encoders
- gguf: https://huggingface.co/Qwen/Qwen3-VL-4B-Instruct-GGUF/tree/main
+24
View File
@@ -0,0 +1,24 @@
# Ming-Image
[Ming-Image](https://github.com/inclusionAI/Ming-Image) 0.1 Design uses a 6B diffusion transformer (DiT), Ling-mini-2.0 for text conditioning, and the Ming-Image VAE. Text-to-image generation with RGBA output is supported.
## Download weights
- Download Ming-Image 0.1 Design DiT
- safetensors: https://huggingface.co/Comfy-Org/Ming-Image/tree/main/diffusion_models
- Download Ling-mini-2.0 BF16
- safetensors: https://huggingface.co/Comfy-Org/Ming-Image/tree/main/text_encoders
- Download Ming-Image VAE
- safetensors: https://huggingface.co/Comfy-Org/Ming-Image/tree/main/vae
- Download Ling tokenizer
- tokenizer.json: https://huggingface.co/inclusionAI/Ming-Image-0.1-Design/blob/main/mllm/tokenizer.json
The example below uses `ming_image_0.1_design_bf16.safetensors` for the DiT. You can also use `ming_image_0.1_design_int8_convrot.safetensors` with [INT8 convrot support](int8_convrot.md). Use the BF16 text encoder.
## Text-to-image
Pass the Ling `tokenizer.json` with `--tokenizer` and use the matching Ming-Image VAE. Image dimensions must be multiples of 16. Save the output as PNG to preserve the alpha channel.
```bash
.\bin\Release\sd-cli.exe --diffusion-model ..\models\diffusion_models\ming_image_0.1_design_bf16.safetensors --llm ..\models\text_encoders\ming_image_0.1_ling_mini_2.0_bf16.safetensors --vae ..\models\vae\ming_image_vae_bf16.safetensors --tokenizer ..\models\text_encoders\tokenizer.json -p "A cheerful orange cat sticker, transparent background" --width 1024 --height 1024 --steps 12 --cfg-scale 1 --sampling-method euler --diffusion-fa -v --offload-to-cpu -o ming_image.png
```
+25
View File
@@ -40,6 +40,9 @@ Wan models require `-M vid_gen`, including single-frame generation. `--video-fra
- safetensors: https://huggingface.co/Comfy-Org/Wan_2.2_ComfyUI_Repackaged/tree/main/split_files/diffusion_models
- gguf: https://huggingface.co/QuantStack/Wan2.2-S2V-14B-GGUF/tree/main
- int8_convrot safetensors: https://huggingface.co/noctrex/Wan2.2-S2V-14B-int8_convrot
- Wan2.2 VACE-Fun A14B
- safetensors: https://huggingface.co/alibaba-pai/Wan2.2-VACE-Fun-A14B
- gguf: https://huggingface.co/QuantStack/Wan2.2-VACE-Fun-A14B-GGUF/tree/main
- Download vae
- wan_2.1_vae (for all the wan model except Wan2.2 TI2V 5B)
- safetensors: https://huggingface.co/Comfy-Org/Wan_2.1_ComfyUI_repackaged/blob/main/split_files/vae/wan_2.1_vae.safetensors
@@ -256,3 +259,25 @@ ffmpeg -i ..\..\ComfyUI\input\post+depth.mp4 -qscale:v 1 -vf fps=8 post+depth\fr
```
<video src=../assets/wan/Wan2.1_14B_vace_v2v.mp4 controls="controls" muted="muted" type="video/mp4"></video>
### Wan2.2 VACE-Fun A14B
VACE-Fun runs as a MoE pair: `--diffusion-model` takes the low-noise expert and
`--high-noise-diffusion-model` the high-noise one. Reference-to-video uses `-i`
for the reference image, same as Wan2.1 VACE.
#### R2V
```
.\bin\Release\sd-cli.exe -M vid_gen --diffusion-model ..\models\diffusion_models\Wan2.2-VACE-Fun-A14B-low-noise-Q8_0.gguf --high-noise-diffusion-model ..\models\diffusion_models\Wan2.2-VACE-Fun-A14B-high-noise-Q8_0.gguf --vae ..\models\vae\wan_2.1_vae.safetensors --t5xxl ..\models\text_encoders\umt5-xxl-encoder-Q8_0.gguf -p "a lovely cat" --cfg-scale 3.5 --sampling-method euler --steps 10 --high-noise-cfg-scale 3.5 --high-noise-sampling-method euler --high-noise-steps 8 -v -n "色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部, 畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走" -W 832 -H 480 --diffusion-fa -i ..\assets\cat_with_sd_cpp_42.png --video-frames 33 --offload-to-cpu
```
<video src=../assets/wan/Wan2.2_A14B_vace_r2v.mp4 controls="controls" muted="muted" type="video/mp4"></video>
#### T2V
Same command without `-i` (VACE context is synthesized from an empty control
video, like Wan2.1 VACE t2v).
> On GPUs with ~12 GB VRAM, VACE also needs `--vae-tiling` — the control-video
> encode can exceed the budget otherwise.
+97 -45
View File
@@ -255,27 +255,48 @@ void refresh_lora_cache(ServerRuntime& rt) {
std::vector<LoraEntry> new_cache;
fs::path lora_dir = rt.ctx_params->lora_model_dir;
if (fs::exists(lora_dir) && fs::is_directory(lora_dir)) {
for (auto& entry : fs::recursive_directory_iterator(lora_dir, fs::directory_options::skip_permission_denied)) {
if (!entry.is_regular_file()) {
continue;
}
const fs::path& p = entry.path();
if (!is_supported_model_ext(p)) {
continue;
}
std::error_code ec;
if (fs::exists(lora_dir, ec) && !ec && fs::is_directory(lora_dir, ec) && !ec) {
try {
auto it = fs::recursive_directory_iterator(
lora_dir,
fs::directory_options::skip_permission_denied,
ec);
auto end = fs::recursive_directory_iterator();
while (!ec && it != end) {
std::error_code entry_ec;
bool is_reg = it->is_regular_file(entry_ec);
if (entry_ec || !is_reg) {
it.increment(ec);
continue;
}
const fs::path& p = it->path();
if (!is_supported_model_ext(p)) {
it.increment(ec);
continue;
}
LoraEntry lora_entry;
lora_entry.name = p.stem().u8string();
lora_entry.fullpath = p.u8string();
std::string rel = p.lexically_relative(lora_dir).u8string();
std::replace(rel.begin(), rel.end(), '\\', '/');
lora_entry.path = rel;
LoraEntry lora_entry;
lora_entry.name = p.stem().u8string();
lora_entry.fullpath = p.u8string();
std::string rel = p.lexically_relative(lora_dir).u8string();
std::replace(rel.begin(), rel.end(), '\\', '/');
lora_entry.path = rel;
new_cache.push_back(std::move(lora_entry));
new_cache.push_back(std::move(lora_entry));
it.increment(ec);
}
} catch (const std::exception& e) {
LOG_WARN("error while scanning lora directory '%s': %s", lora_dir.string().c_str(), e.what());
return;
}
}
if (ec) {
LOG_WARN("error while scanning lora directory '%s': %s", lora_dir.string().c_str(), ec.message().c_str());
return;
}
std::sort(new_cache.begin(), new_cache.end(), [](const LoraEntry& a, const LoraEntry& b) {
return a.path < b.path;
});
@@ -302,40 +323,71 @@ void refresh_upscaler_cache(ServerRuntime& rt) {
}
fs::path upscaler_dir = rt.ctx_params->hires_upscalers_dir;
if (fs::exists(upscaler_dir) && fs::is_directory(upscaler_dir)) {
for (auto& entry : fs::directory_iterator(upscaler_dir)) {
if (!entry.is_regular_file()) {
continue;
}
const fs::path& p = entry.path();
if (!is_supported_model_ext(p)) {
continue;
}
std::error_code ec;
if (fs::exists(upscaler_dir, ec) && !ec && fs::is_directory(upscaler_dir, ec) && !ec) {
try {
auto it = fs::directory_iterator(
upscaler_dir,
fs::directory_options::skip_permission_denied,
ec);
auto end = fs::directory_iterator();
while (!ec && it != end) {
std::error_code entry_ec;
bool is_reg = it->is_regular_file(entry_ec);
if (entry_ec || !is_reg) {
it.increment(ec);
continue;
}
const fs::path& p = it->path();
if (!is_supported_model_ext(p)) {
it.increment(ec);
continue;
}
UpscalerEntry upscaler_entry;
upscaler_entry.name = p.stem().u8string();
upscaler_entry.fullpath = fs::absolute(p).lexically_normal().u8string();
upscaler_entry.model_name = "ESRGAN_4x";
upscaler_entry.path = p.filename().u8string();
upscaler_entry.file_size = entry.file_size();
upscaler_entry.last_modified = entry.last_write_time();
auto previous = std::find_if(previous_cache.begin(), previous_cache.end(), [&](const UpscalerEntry& cached) {
return cached.fullpath == upscaler_entry.fullpath &&
cached.file_size == upscaler_entry.file_size &&
cached.last_modified == upscaler_entry.last_modified;
});
upscaler_entry.image_upscale_factor = previous != previous_cache.end()
? previous->image_upscale_factor
: get_upscaler_model_scale(upscaler_entry.fullpath.c_str());
if (upscaler_entry.image_upscale_factor > 0) {
upscaler_entry.scale = upscaler_entry.image_upscale_factor;
upscaler_entry.model_name = "ESRGAN_" + std::to_string(upscaler_entry.scale) + "x";
}
UpscalerEntry upscaler_entry;
upscaler_entry.name = p.stem().u8string();
std::error_code abs_ec;
fs::path abs_path = fs::absolute(p, abs_ec);
upscaler_entry.fullpath = (abs_ec ? p : abs_path.lexically_normal()).u8string();
upscaler_entry.model_name = "ESRGAN_4x";
upscaler_entry.path = p.filename().u8string();
new_cache.push_back(std::move(upscaler_entry));
std::error_code size_ec;
upscaler_entry.file_size = it->file_size(size_ec);
if (size_ec) {
it.increment(ec);
continue;
}
std::error_code time_ec;
upscaler_entry.last_modified = it->last_write_time(time_ec);
auto previous = std::find_if(previous_cache.begin(), previous_cache.end(), [&](const UpscalerEntry& cached) {
return cached.fullpath == upscaler_entry.fullpath &&
cached.file_size == upscaler_entry.file_size &&
(!time_ec && cached.last_modified == upscaler_entry.last_modified);
});
upscaler_entry.image_upscale_factor = previous != previous_cache.end()
? previous->image_upscale_factor
: get_upscaler_model_scale(upscaler_entry.fullpath.c_str());
if (upscaler_entry.image_upscale_factor > 0) {
upscaler_entry.scale = upscaler_entry.image_upscale_factor;
upscaler_entry.model_name = "ESRGAN_" + std::to_string(upscaler_entry.scale) + "x";
}
new_cache.push_back(std::move(upscaler_entry));
it.increment(ec);
}
} catch (const std::exception& e) {
LOG_WARN("error while scanning upscalers directory '%s': %s", upscaler_dir.string().c_str(), e.what());
return;
}
}
if (ec) {
LOG_WARN("error while scanning upscalers directory '%s': %s", upscaler_dir.string().c_str(), ec.message().c_str());
return;
}
std::sort(new_cache.begin(), new_cache.end(), [](const UpscalerEntry& a, const UpscalerEntry& b) {
return a.name < b.name;
});
+1 -1
Submodule ggml updated: 4bf5f60006...610265a4b0
+60
View File
@@ -16,6 +16,7 @@
#include "model/te/clip.hpp"
#include "model/te/llada_image_te.hpp"
#include "model/te/llm.hpp"
#include "model/te/ming_image_te.hpp"
#include "model/te/t5.hpp"
#include "model_loader.h"
#include "tokenizers/sensenova_u1_tokenizer.h"
@@ -3171,6 +3172,65 @@ struct LLMEmbedder : public Conditioner {
}
};
struct MingImageEmbedder : public Conditioner {
std::shared_ptr<Tokenizer> tokenizer;
std::shared_ptr<MingImageTE::MingImageTextRunner> text_model;
const std::string prefix = "text_encoders.llm";
MingImageEmbedder(ggml_backend_t backend, const String2TensorStorage& tensors, std::shared_ptr<RunnerWeightManager> weight_manager, const TokenizerConfig& tokenizers) {
if (!tokenizers.has(TokenizerConfig::MAIN)) {
throw std::runtime_error("Ming-Image requires the Ling tokenizer.json; pass --tokenizer FILE");
}
text_model = std::make_shared<MingImageTE::MingImageTextRunner>(backend, tensors, prefix, weight_manager);
tokenizer = tokenizers.create(TokenizerConfig::MAIN, text_model->config.backbone.vocab_size, 156895);
}
void get_param_tensors(std::map<std::string, ggml_tensor*>& tensors) override {
text_model->get_param_tensors(tensors, prefix);
}
void get_param_tensor_ops(std::map<ggml_tensor*, enum ggml_op>& ops) override {
text_model->get_param_tensor_ops(ops);
}
void get_layer_split_param_tensors(std::map<std::string, ggml_tensor*>& tensors) override {
text_model->get_param_tensors(tensors, prefix);
}
void set_flash_attention_enabled(bool enabled) override { text_model->set_flash_attention_enabled(enabled); }
void set_max_graph_vram_bytes(size_t bytes) override { text_model->set_max_graph_vram_bytes(bytes); }
void set_runtime_backends(const std::vector<ggml_backend_t>& backends) override { text_model->set_runtime_backends(backends); }
void set_graph_cut_layer_split_enabled(bool enabled) override { text_model->set_graph_cut_layer_split_enabled(enabled); }
void set_graph_cut_layer_split_backend_vram_limits(const std::vector<size_t>& limits) override { text_model->set_graph_cut_layer_split_backend_vram_limits(limits); }
void set_scale_overrides(float linear, float attention) override { text_model->set_scale_overrides(linear, attention); }
void set_weight_adapter(const std::shared_ptr<WeightAdapter>& adapter) override { text_model->set_weight_adapter(adapter); }
void runner_end() override { text_model->runner_end(); }
SDCondition get_learned_condition(int n_threads, const ConditionerParams& input) override {
if (input.ref_images != nullptr && !input.ref_images->empty()) {
LOG_ERROR("Ming-Image currently supports text-to-image only");
return {};
}
std::string prompt =
"<role>SYSTEM</role>你是一个友好的AI助手。\n\ndetailed thinking off<|role_end|>"
"<role>HUMAN</role>" +
input.text + "<|role_end|><role>ASSISTANT</role>";
std::vector<int> tokens;
if (!tokenizer->encode(prompt, tokens, nullptr)) {
return {};
}
if (tokens.size() + 258 > 32768) {
LOG_ERROR("Ming-Image prompt exceeds the text encoder context length");
return {};
}
auto output = text_model->compute(n_threads, tokens);
if (output.empty()) {
return {};
}
SDCondition result;
result.c_crossattn = sd::ops::slice(sd::ops::slice(output, 1, 0, 256), 0, 0, text_model->config.caption_dim);
result.extra_c_crossattns.push_back(sd::ops::slice(output, 1, 256, output.shape()[1]));
return result;
}
};
struct LTXAVTextProjection : public GGMLBlock {
static constexpr int64_t kHiddenSize = 3840;
static constexpr int64_t kNumStates = 49;
+3 -1
View File
@@ -63,6 +63,7 @@ enum SDVersion {
VERSION_LLADA_IMAGE,
VERSION_ESRGAN,
VERSION_PIXART,
VERSION_MING_IMAGE,
VERSION_COUNT,
};
@@ -280,7 +281,7 @@ static inline bool sd_version_uses_flux2_vae(SDVersion version) {
}
static inline bool sd_version_uses_wan_vae(SDVersion version) {
if (sd_version_is_wan(version) || sd_version_is_lingbot_video(version) || sd_version_is_qwen_image(version) || sd_version_is_krea2(version) || sd_version_is_anima(version)) {
if (sd_version_is_wan(version) || sd_version_is_lingbot_video(version) || sd_version_is_qwen_image(version) || sd_version_is_krea2(version) || sd_version_is_anima(version) || version == VERSION_MING_IMAGE) {
return true;
}
return false;
@@ -314,6 +315,7 @@ static inline bool sd_version_is_dit(SDVersion version) {
version == VERSION_HIDREAM_O1 ||
sd_version_is_anima(version) ||
sd_version_is_z_image(version) ||
version == VERSION_MING_IMAGE ||
sd_version_is_llada_image(version) ||
sd_version_is_boogu_image(version) ||
sd_version_is_ernie_image(version) ||
+75 -6
View File
@@ -1,6 +1,7 @@
#ifndef __SD_MODEL_COMMON_GGML_BLOCK_HPP__
#define __SD_MODEL_COMMON_GGML_BLOCK_HPP__
#include <cmath>
#include <cstdint>
#include <map>
#include <memory>
@@ -214,7 +215,7 @@ public:
if (ctx->backend != nullptr) {
ggml_tensor* fp8_matmul = ggml_mul_mat(ctx->ggml_ctx, w, x);
if (force_prec_f32) {
ggml_mul_mat_set_prec(fp8_matmul, GGML_PREC_F32);
ggml_prec_set_acc(fp8_matmul, GGML_PREC_F32);
}
supports_fp8_matmul = ggml_backend_supports_op(ctx->backend, fp8_matmul);
}
@@ -335,15 +336,75 @@ class Embedding : public UnaryBlock {
protected:
int64_t embedding_dim;
int64_t num_embeddings;
bool is_int8_tensorwise = false;
int int8_convrot_group_size = 0;
static void get_rows_i8(ggml_tensor* dst, int ith, int nth, void* userdata) {
const auto* embedding = static_cast<const Embedding*>(userdata);
const int group_size = embedding->int8_convrot_group_size;
const auto* weight = dst->src[0];
const auto* input_ids = static_cast<const int32_t*>(dst->src[1]->data);
const auto* scales = static_cast<const float*>(dst->src[2]->data);
const bool scalar_scale = ggml_nelements(dst->src[2]) == 1;
const float normalization = group_size > 0 ? 1.f / std::sqrt(static_cast<float>(group_size)) : 1.f;
for (int64_t row = ith; row < dst->ne[1]; row += nth) {
const int32_t token = input_ids[row];
GGML_ASSERT(token >= 0 && token < weight->ne[1]);
const auto* src = reinterpret_cast<const int8_t*>(static_cast<const char*>(weight->data) + token * weight->nb[1]);
auto* out = reinterpret_cast<float*>(static_cast<char*>(dst->data) + row * dst->nb[1]);
const float scale = scales[scalar_scale ? 0 : token] * normalization;
for (int64_t i = 0; i < weight->ne[0]; ++i) {
out[i] = static_cast<float>(src[i]) * scale;
}
// The regular Hadamard rotation is symmetric and its own inverse.
for (int stride = 1; stride < group_size; stride *= 4) {
for (int64_t base = 0; base < weight->ne[0]; base += 4 * stride) {
for (int j = 0; j < stride; ++j) {
float* values = out + base + j;
const float a = values[0];
const float b = values[stride];
const float c = values[2 * stride];
const float d = values[3 * stride];
values[0] = a + b + c - d;
values[stride] = a + b - c + d;
values[2 * stride] = a - b + c + d;
values[3 * stride] = -a + b + c + d;
}
}
}
}
}
void init_params(ggml_context* ctx, const String2TensorStorage& tensor_storage_map, const std::string prefix = "") override {
enum ggml_type wtype = get_type(prefix + "weight", tensor_storage_map, GGML_TYPE_F32);
if (!support_get_rows(wtype)) {
auto weight_storage = tensor_storage_map.find(prefix + "weight");
is_int8_tensorwise = weight_storage != tensor_storage_map.end() && weight_storage->second.is_int8_tensorwise;
int8_convrot_group_size = 0;
enum ggml_type wtype = get_type(prefix + "weight", tensor_storage_map, GGML_TYPE_F32);
if (is_int8_tensorwise) {
GGML_ASSERT(wtype == GGML_TYPE_I8);
auto scale_storage = tensor_storage_map.find(prefix + "weight_scale");
GGML_ASSERT(scale_storage != tensor_storage_map.end());
const int64_t scale_nelements = scale_storage->second.nelements();
GGML_ASSERT(scale_nelements == 1 || scale_nelements == num_embeddings);
params["weight_scale"] = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, scale_nelements);
if (weight_storage->second.int8_convrot) {
int8_convrot_group_size = weight_storage->second.int8_convrot_group_size;
int remainder = int8_convrot_group_size;
while (remainder > 1 && remainder % 4 == 0) {
remainder /= 4;
}
GGML_ASSERT(remainder == 1 && embedding_dim % int8_convrot_group_size == 0);
}
} else if (!support_get_rows(wtype)) {
wtype = GGML_TYPE_F32;
}
params["weight"] = ggml_new_tensor_2d(ctx, wtype, embedding_dim, num_embeddings);
}
enum ggml_op param_usage_op(const std::string& name) const override {
if (is_int8_tensorwise) {
return GGML_OP_CUSTOM;
}
return name == "weight" ? GGML_OP_GET_ROWS : GGML_OP_NONE;
}
@@ -363,9 +424,17 @@ public:
int64_t n = input_ids->ne[1];
input_ids = ggml_reshape_1d(ctx->ggml_ctx, input_ids, input_ids->ne[0] * input_ids->ne[1]);
input_ids = ggml_reshape_3d(ctx->ggml_ctx, input_ids, input_ids->ne[0], 1, input_ids->ne[1]);
auto embedding = ggml_get_rows(ctx->ggml_ctx, weight, input_ids);
embedding = ggml_reshape_3d(ctx->ggml_ctx, embedding, embedding->ne[0], embedding->ne[1] / n, n);
ggml_tensor* embedding;
if (is_int8_tensorwise) {
GGML_ASSERT(input_ids->type == GGML_TYPE_I32);
ggml_tensor* args[] = {weight, input_ids, params["weight_scale"]};
embedding = ggml_custom_4d(ctx->ggml_ctx, GGML_TYPE_F32, embedding_dim, input_ids->ne[0], 1, 1,
args, 3, get_rows_i8, GGML_N_TASKS_MAX, this);
} else {
input_ids = ggml_reshape_3d(ctx->ggml_ctx, input_ids, input_ids->ne[0], 1, input_ids->ne[1]);
embedding = ggml_get_rows(ctx->ggml_ctx, weight, input_ids);
}
embedding = ggml_reshape_3d(ctx->ggml_ctx, embedding, embedding->ne[0], embedding->ne[1] / n, n);
// [N, n_token, embedding_dim]
return embedding;
+135
View File
@@ -0,0 +1,135 @@
#ifndef __SD_MODEL_DIFFUSION_MING_IMAGE_HPP__
#define __SD_MODEL_DIFFUSION_MING_IMAGE_HPP__
#include "z_image.hpp"
namespace MingImage {
struct MingImageConfig : ZImage::ZImageConfig {
bool split_qkv = true;
static MingImageConfig detect_from_weights(const String2TensorStorage& tensors, const std::string& prefix) {
MingImageConfig config;
static_cast<ZImage::ZImageConfig&>(config) = ZImage::ZImageConfig::detect_from_weights(tensors, prefix);
config.split_qkv = tensors.count(prefix + ".layers.0.attention.qkv.weight") == 0;
return config;
}
};
class MingImageModel : public GGMLBlock {
MingImageConfig config;
public:
explicit MingImageModel(const MingImageConfig& config)
: config(config) {
blocks["x_embedder"] = std::make_shared<Linear>(config.patch_size * config.patch_size * config.in_channels, config.hidden_size);
blocks["t_embedder"] = std::make_shared<TimestepEmbedder>(1024, 256, std::min<int64_t>(config.hidden_size, 256));
blocks["cap_embedder.0"] = std::make_shared<RMSNorm>(config.cap_feat_dim, config.norm_eps);
blocks["cap_embedder.1"] = std::make_shared<Linear>(config.cap_feat_dim, config.hidden_size);
auto add_blocks = [&](const std::string& prefix, int64_t count, bool modulation) {
for (int64_t i = 0; i < count; ++i) {
blocks[prefix + std::to_string(i)] = std::make_shared<ZImage::JointTransformerBlock>(
static_cast<int>(i), config.hidden_size, config.head_dim, config.num_heads,
config.num_kv_heads, config.multiple_of, config.ffn_dim_multiplier,
config.norm_eps, config.qk_norm, modulation, true, config.split_qkv, 1e-5f);
}
};
add_blocks("noise_refiner.", config.num_refiner_layers, true);
add_blocks("context_refiner.", config.num_refiner_layers, false);
add_blocks("layers.", config.num_layers, true);
blocks["final_layer"] = std::make_shared<ZImage::FinalLayer>(config.hidden_size, config.patch_size, config.out_channels);
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x, ggml_tensor* timestep, ggml_tensor* context, ggml_tensor* direct, ggml_tensor* pe) {
auto gctx = ctx->ggml_ctx;
const int64_t width = x->ne[0], height = x->ne[1];
auto img = DiT::pad_and_patchify(ctx, x, config.patch_size, config.patch_size, false);
img = std::dynamic_pointer_cast<Linear>(blocks["x_embedder"])->forward(ctx, img);
auto txt = std::dynamic_pointer_cast<RMSNorm>(blocks["cap_embedder.0"])->forward(ctx, context);
txt = std::dynamic_pointer_cast<Linear>(blocks["cap_embedder.1"])->forward(ctx, txt);
txt = ggml_concat(gctx, txt, direct, 1);
auto t = std::dynamic_pointer_cast<TimestepEmbedder>(blocks["t_embedder"])->forward(ctx, timestep);
const int64_t n_txt = txt->ne[1], n_img = img->ne[1];
auto txt_pe = ggml_ext_slice(gctx, pe, 3, 0, n_txt);
auto img_pe = ggml_ext_slice(gctx, pe, 3, n_txt, n_txt + n_img);
for (int64_t i = 0; i < config.num_refiner_layers; ++i) {
txt = std::dynamic_pointer_cast<ZImage::JointTransformerBlock>(blocks["context_refiner." + std::to_string(i)])->forward(ctx, txt, txt_pe);
sd::ggml_graph_cut::mark_graph_cut(txt, "ming_image.context_refiner." + std::to_string(i), "txt");
}
for (int64_t i = 0; i < config.num_refiner_layers; ++i) {
img = std::dynamic_pointer_cast<ZImage::JointTransformerBlock>(blocks["noise_refiner." + std::to_string(i)])->forward(ctx, img, img_pe, nullptr, t);
sd::ggml_graph_cut::mark_graph_cut(img, "ming_image.noise_refiner." + std::to_string(i), "img");
}
auto combined = ggml_concat(gctx, txt, img, 1);
for (int64_t i = 0; i < config.num_layers; ++i) {
combined = std::dynamic_pointer_cast<ZImage::JointTransformerBlock>(blocks["layers." + std::to_string(i)])->forward(ctx, combined, pe, nullptr, t);
sd::ggml_graph_cut::mark_graph_cut(combined, "ming_image.layers." + std::to_string(i), "combined");
}
img = ggml_ext_slice(gctx, combined, 1, n_txt, n_txt + n_img);
img = std::dynamic_pointer_cast<ZImage::FinalLayer>(blocks["final_layer"])->forward(ctx, img, t);
img = DiT::unpatchify_and_crop(gctx, img, height, width, config.patch_size, config.patch_size, false);
return ggml_scale(gctx, img, -1.f);
}
};
struct MingImageRunner : DiffusionModelRunner {
MingImageConfig config;
MingImageModel model;
std::vector<float> pe_values;
MingImageRunner(ggml_backend_t backend, const String2TensorStorage& tensors, const std::string& prefix, std::shared_ptr<RunnerWeightManager> weight_manager = nullptr)
: DiffusionModelRunner(backend, prefix, weight_manager),
config(MingImageConfig::detect_from_weights(tensors, prefix)),
model(config) {
model.init(params_ctx, tensors, prefix);
}
std::string get_desc() override { return "ming_image"; }
void get_param_tensors(std::map<std::string, ggml_tensor*>& tensors, const std::string& prefix) override {
model.get_param_tensors(tensors, prefix);
}
sd::Tensor<float> compute(int n_threads, const DiffusionParams& inputs) override {
const auto* extra = diffusion_extra_as<MingImageDiffusionExtra>(inputs);
if (inputs.ref_latents != nullptr && !inputs.ref_latents->empty()) {
LOG_ERROR("Ming-Image reference-image conditioning is not supported");
return {};
}
if (inputs.context == nullptr || extra->direct_context == nullptr) {
LOG_ERROR("Ming-Image requires both query and direct text conditions");
return {};
}
auto graph = [&]() {
auto gf = new_graph_custom(ZImage::Z_IMAGE_GRAPH_SIZE);
auto x = make_input(*inputs.x);
auto t = make_input(*inputs.timesteps);
auto context = make_input(*inputs.context);
auto direct = make_input(*extra->direct_context);
GGML_ASSERT(x->ne[3] == 1);
const int64_t n_txt = context->ne[1] + direct->ne[1];
const int64_t n_img = ((x->ne[0] + config.patch_size - 1) / config.patch_size) *
((x->ne[1] + config.patch_size - 1) / config.patch_size);
auto padded = finish_rope_pe(Rope::gen_z_image_pe(
static_cast<int>(x->ne[1]), static_cast<int>(x->ne[0]), config.patch_size, 1,
static_cast<int>(n_txt), ZImage::SEQ_MULTI_OF, {}, Rope::RefIndexMode::FIXED,
config.theta, config.axes_dim));
// Zero-masked alignment tokens cannot affect valid queries. Omit them while
// retaining the padded caption length used to position image tokens.
const size_t stride = config.axes_dim_sum * 2;
const int64_t padded_txt = n_txt + Rope::bound_mod(static_cast<int>(n_txt), ZImage::SEQ_MULTI_OF);
pe_values.assign(padded.begin(), padded.begin() + n_txt * stride);
pe_values.insert(pe_values.end(), padded.begin() + padded_txt * stride,
padded.begin() + (padded_txt + n_img) * stride);
auto pe = ggml_new_tensor_4d(compute_ctx, GGML_TYPE_F32, 2, 2, config.axes_dim_sum / 2, n_txt + n_img);
set_backend_tensor_data(pe, pe_values.data());
auto ctx = get_context();
auto out = model.forward(&ctx, x, t, context, direct, pe);
ggml_build_forward_expand(gf, out);
return gf;
};
return restore_trailing_singleton_dims(GGMLRunner::compute(graph, n_threads, false), inputs.x->dim());
}
};
}
#endif // __SD_MODEL_DIFFUSION_MING_IMAGE_HPP__
+6 -1
View File
@@ -141,6 +141,10 @@ struct LLaDAImageDiffusionExtra {
const sd::Tensor<float>* semantic = nullptr;
};
struct MingImageDiffusionExtra {
const sd::Tensor<float>* direct_context = nullptr;
};
using DiffusionExtraParams = std::variant<std::monostate,
UNetDiffusionExtra,
SkipLayerDiffusionExtra,
@@ -154,7 +158,8 @@ using DiffusionExtraParams = std::variant<std::monostate,
MiniT2IDiffusionExtra,
SenseNovaU1DiffusionExtra,
HunyuanVideoDiffusionExtra,
LLaDAImageDiffusionExtra>;
LLaDAImageDiffusionExtra,
MingImageDiffusionExtra>;
struct DiffusionParams {
const sd::Tensor<float>* x = nullptr;
+70 -61
View File
@@ -41,28 +41,28 @@ namespace PixArt {
auto it = weights.find(prefix + "." + suffix);
return it == weights.end() ? nullptr : &it->second;
};
if (auto w = find("pos_embed.proj.weight")) {
if (auto w = find("x_embedder.proj.weight")) {
config.hidden_size = w->ne[3];
config.in_channels = w->ne[2];
config.patch_size = w->ne[0];
}
if (auto w = find("proj_out.weight")) {
if (auto w = find("final_layer.linear.weight")) {
config.out_channels = w->ne[1] / (config.patch_size * config.patch_size);
}
if (auto w = find("caption_projection.linear_1.weight")) {
if (auto w = find("y_embedder.y_proj.fc1.weight")) {
config.caption_channels = w->ne[0];
}
if (auto w = find("transformer_blocks.0.attn2.to_k.weight")) {
if (auto w = find("blocks.0.cross_attn.kv_linear.weight")) {
config.cross_attention_dim = w->ne[0];
}
if (auto w = find("transformer_blocks.0.ff.net.0.proj.weight")) {
if (auto w = find("blocks.0.mlp.fc1.weight")) {
config.ffn_dim = w->ne[1];
}
if (find("adaln_single.emb.resolution_embedder.linear_1.weight") != nullptr) {
if (find("csize_embedder.mlp.0.weight") != nullptr) {
LOG_WARN("pixart: resolution/aspect-ratio micro conditions are not supported; output may differ from the reference");
}
int layers = 0;
const std::string block_prefix = prefix + ".transformer_blocks.";
const std::string block_prefix = prefix + ".blocks.";
for (const auto& [name, _] : weights) {
if (starts_with(name, block_prefix)) {
layers = std::max(layers, atoi(name.substr(block_prefix.size()).c_str()) + 1);
@@ -108,37 +108,46 @@ namespace PixArt {
class PixArtTimestepEmbedding : public GGMLBlock {
public:
PixArtTimestepEmbedding(int64_t in_channels, int64_t out_dim) {
blocks["linear_1"] = std::make_shared<Linear>(in_channels, out_dim);
blocks["linear_2"] = std::make_shared<Linear>(out_dim, out_dim);
blocks["mlp.0"] = std::make_shared<Linear>(in_channels, out_dim);
blocks["mlp.2"] = std::make_shared<Linear>(out_dim, out_dim);
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) {
x = std::dynamic_pointer_cast<Linear>(blocks["linear_1"])->forward(ctx, x);
x = std::dynamic_pointer_cast<Linear>(blocks["mlp.0"])->forward(ctx, x);
x = ggml_silu(ctx->ggml_ctx, x);
return std::dynamic_pointer_cast<Linear>(blocks["linear_2"])->forward(ctx, x);
return std::dynamic_pointer_cast<Linear>(blocks["mlp.2"])->forward(ctx, x);
}
};
class PixArtAttention : public GGMLBlock {
int64_t num_heads;
bool self_attention;
public:
PixArtAttention(int64_t dim, int64_t num_heads, int64_t context_dim)
: num_heads(num_heads) {
blocks["to_q"] = std::make_shared<Linear>(dim, dim);
blocks["to_k"] = std::make_shared<Linear>(context_dim, dim);
blocks["to_v"] = std::make_shared<Linear>(context_dim, dim);
blocks["to_out.0"] = std::make_shared<Linear>(dim, dim);
PixArtAttention(int64_t dim, int64_t num_heads, int64_t context_dim, bool self_attention)
: num_heads(num_heads), self_attention(self_attention) {
if (self_attention) {
blocks["qkv"] = std::make_shared<Linear>(dim, 3 * dim);
} else {
blocks["q_linear"] = std::make_shared<Linear>(dim, dim);
blocks["kv_linear"] = std::make_shared<Linear>(context_dim, 2 * dim);
}
blocks["proj"] = std::make_shared<Linear>(dim, dim);
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x, ggml_tensor* context, ggml_tensor* mask = nullptr) {
// x: [N, n_token, dim], context: [N, n_context, context_dim]
auto q = std::dynamic_pointer_cast<Linear>(blocks["to_q"])->forward(ctx, x);
auto k = std::dynamic_pointer_cast<Linear>(blocks["to_k"])->forward(ctx, context);
auto v = std::dynamic_pointer_cast<Linear>(blocks["to_v"])->forward(ctx, context);
auto out = ggml_ext_attention_ext(ctx, q, k, v, num_heads, mask, false, ctx->flash_attn_enabled);
return std::dynamic_pointer_cast<Linear>(blocks["to_out.0"])->forward(ctx, out);
std::vector<ggml_tensor*> qkv;
if (self_attention) {
auto projected = std::dynamic_pointer_cast<Linear>(blocks["qkv"])->forward(ctx, x);
qkv = ggml_ext_chunk(ctx->ggml_ctx, projected, 3, 0);
} else {
auto q = std::dynamic_pointer_cast<Linear>(blocks["q_linear"])->forward(ctx, x);
auto kv = std::dynamic_pointer_cast<Linear>(blocks["kv_linear"])->forward(ctx, context);
auto parts = ggml_ext_chunk(ctx->ggml_ctx, kv, 2, 0);
qkv = {q, parts[0], parts[1]};
}
auto out = ggml_ext_attention_ext(ctx, qkv[0], qkv[1], qkv[2], num_heads, mask, false, ctx->flash_attn_enabled);
return std::dynamic_pointer_cast<Linear>(blocks["proj"])->forward(ctx, out);
}
};
@@ -155,10 +164,10 @@ namespace PixArt {
public:
PixArtBlock(int64_t dim, int64_t num_heads, int64_t context_dim, int64_t ffn_dim)
: dim(dim) {
blocks["attn1"] = std::make_shared<PixArtAttention>(dim, num_heads, dim);
blocks["attn2"] = std::make_shared<PixArtAttention>(dim, num_heads, context_dim);
blocks["ff.net.0.proj"] = std::make_shared<Linear>(dim, ffn_dim);
blocks["ff.net.2"] = std::make_shared<Linear>(ffn_dim, dim);
blocks["attn"] = std::make_shared<PixArtAttention>(dim, num_heads, dim, true);
blocks["cross_attn"] = std::make_shared<PixArtAttention>(dim, num_heads, context_dim, false);
blocks["mlp.fc1"] = std::make_shared<Linear>(dim, ffn_dim);
blocks["mlp.fc2"] = std::make_shared<Linear>(ffn_dim, dim);
}
static ggml_tensor* norm(ggml_context* ctx, ggml_tensor* x) {
@@ -174,14 +183,14 @@ namespace PixArt {
if (table->type != GGML_TYPE_F32) {
table = ggml_cast(ctx->ggml_ctx, table, GGML_TYPE_F32);
}
table = ggml_reshape_3d(ctx->ggml_ctx, table, dim, 6, 1);
auto m = ggml_add(ctx->ggml_ctx, ggml_reshape_3d(ctx->ggml_ctx, mod, dim, 6, N), table);
auto mv = ggml_ext_chunk(ctx->ggml_ctx, ggml_reshape_2d(ctx->ggml_ctx, ggml_ext_cont(ctx->ggml_ctx, m), dim * 6, N), 6, 0);
table = ggml_reshape_3d(ctx->ggml_ctx, table, dim, 6, 1);
auto m = ggml_add(ctx->ggml_ctx, ggml_reshape_3d(ctx->ggml_ctx, mod, dim, 6, N), table);
auto mv = ggml_ext_chunk(ctx->ggml_ctx, ggml_reshape_2d(ctx->ggml_ctx, ggml_ext_cont(ctx->ggml_ctx, m), dim * 6, N), 6, 0);
auto attn1 = std::dynamic_pointer_cast<PixArtAttention>(blocks["attn1"]);
auto attn2 = std::dynamic_pointer_cast<PixArtAttention>(blocks["attn2"]);
auto proj = std::dynamic_pointer_cast<Linear>(blocks["ff.net.0.proj"]);
auto fc2 = std::dynamic_pointer_cast<Linear>(blocks["ff.net.2"]);
auto attn1 = std::dynamic_pointer_cast<PixArtAttention>(blocks["attn"]);
auto attn2 = std::dynamic_pointer_cast<PixArtAttention>(blocks["cross_attn"]);
auto proj = std::dynamic_pointer_cast<Linear>(blocks["mlp.fc1"]);
auto fc2 = std::dynamic_pointer_cast<Linear>(blocks["mlp.fc2"]);
auto gate = [&](ggml_tensor* y, ggml_tensor* g) {
g = ggml_reshape_3d(ctx->ggml_ctx, g, dim, 1, N);
@@ -206,29 +215,29 @@ namespace PixArt {
void init_params(ggml_context* ctx,
const String2TensorStorage& tensor_storage_map = {},
const std::string prefix = "") override {
ggml_type wtype = get_type(prefix + "scale_shift_table", tensor_storage_map, GGML_TYPE_F32);
params["scale_shift_table"] = ggml_new_tensor_2d(ctx, wtype, config.hidden_size, 2);
ggml_type wtype = get_type(prefix + "final_layer.scale_shift_table", tensor_storage_map, GGML_TYPE_F32);
params["final_layer.scale_shift_table"] = ggml_new_tensor_2d(ctx, wtype, config.hidden_size, 2);
}
public:
PixArtModel() = default;
PixArtModel(const PixArtConfig& config)
: config(config) {
blocks["pos_embed.proj"] = std::make_shared<Conv2d>(config.in_channels,
config.hidden_size,
std::pair<int, int>{static_cast<int>(config.patch_size), static_cast<int>(config.patch_size)},
std::pair<int, int>{static_cast<int>(config.patch_size), static_cast<int>(config.patch_size)});
blocks["adaln_single.emb.timestep_embedder"] = std::make_shared<PixArtTimestepEmbedding>(ADALN_EMBED_DIM, config.hidden_size);
blocks["adaln_single.linear"] = std::make_shared<Linear>(config.hidden_size, 6 * config.hidden_size);
blocks["caption_projection.linear_1"] = std::make_shared<Linear>(config.caption_channels, config.hidden_size);
blocks["caption_projection.linear_2"] = std::make_shared<Linear>(config.hidden_size, config.cross_attention_dim);
blocks["x_embedder.proj"] = std::make_shared<Conv2d>(config.in_channels,
config.hidden_size,
std::pair<int, int>{static_cast<int>(config.patch_size), static_cast<int>(config.patch_size)},
std::pair<int, int>{static_cast<int>(config.patch_size), static_cast<int>(config.patch_size)});
blocks["t_embedder"] = std::make_shared<PixArtTimestepEmbedding>(ADALN_EMBED_DIM, config.hidden_size);
blocks["t_block.1"] = std::make_shared<Linear>(config.hidden_size, 6 * config.hidden_size);
blocks["y_embedder.y_proj.fc1"] = std::make_shared<Linear>(config.caption_channels, config.hidden_size);
blocks["y_embedder.y_proj.fc2"] = std::make_shared<Linear>(config.hidden_size, config.cross_attention_dim);
for (int i = 0; i < config.num_layers; ++i) {
blocks["transformer_blocks." + std::to_string(i)] =
blocks["blocks." + std::to_string(i)] =
std::make_shared<PixArtBlock>(config.hidden_size, config.num_heads, config.cross_attention_dim, config.ffn_dim);
}
blocks["norm_out"] = std::make_shared<LayerNorm>(config.hidden_size, 1e-6f, false);
blocks["proj_out"] = std::make_shared<Linear>(config.hidden_size,
config.patch_size * config.patch_size * config.out_channels);
blocks["final_layer.norm_final"] = std::make_shared<LayerNorm>(config.hidden_size, 1e-6f, false);
blocks["final_layer.linear"] = std::make_shared<Linear>(config.hidden_size,
config.patch_size * config.patch_size * config.out_channels);
}
ggml_tensor* forward(GGMLRunnerContext* ctx,
@@ -245,29 +254,29 @@ namespace PixArt {
int64_t wp = W / p;
int64_t hp = H / p;
auto h = std::dynamic_pointer_cast<Conv2d>(blocks["pos_embed.proj"])->forward(ctx, x); // [N, hidden, hp, wp]
h = ggml_ext_cont(ctx->ggml_ctx, ggml_permute(ctx->ggml_ctx, h, 1, 2, 0, 3)); // [N, hp, wp, hidden] -> [N, hp*wp, hidden]
h = ggml_reshape_3d(ctx->ggml_ctx, h, config.hidden_size, wp * hp, N); // [N, hp*wp, hidden]
auto h = std::dynamic_pointer_cast<Conv2d>(blocks["x_embedder.proj"])->forward(ctx, x); // [N, hidden, hp, wp]
h = ggml_ext_cont(ctx->ggml_ctx, ggml_permute(ctx->ggml_ctx, h, 1, 2, 0, 3)); // [N, hp, wp, hidden] -> [N, hp*wp, hidden]
h = ggml_reshape_3d(ctx->ggml_ctx, h, config.hidden_size, wp * hp, N); // [N, hp*wp, hidden]
h = ggml_add(ctx->ggml_ctx, h, pos_embed);
auto t = ggml_ext_timestep_embedding(ctx->ggml_ctx, timesteps, ADALN_EMBED_DIM, 10000);
auto emb = std::dynamic_pointer_cast<PixArtTimestepEmbedding>(blocks["adaln_single.emb.timestep_embedder"])->forward(ctx, t);
auto emb = std::dynamic_pointer_cast<PixArtTimestepEmbedding>(blocks["t_embedder"])->forward(ctx, t);
auto mod = std::dynamic_pointer_cast<Linear>(blocks["adaln_single.linear"])
auto mod = std::dynamic_pointer_cast<Linear>(blocks["t_block.1"])
->forward(ctx, ggml_silu(ctx->ggml_ctx, emb)); // [N, 6 * hidden]
auto ctx_emb = std::dynamic_pointer_cast<Linear>(blocks["caption_projection.linear_1"])->forward(ctx, context);
auto ctx_emb = std::dynamic_pointer_cast<Linear>(blocks["y_embedder.y_proj.fc1"])->forward(ctx, context);
ctx_emb = ggml_ext_gelu(ctx->ggml_ctx, ctx_emb, true);
ctx_emb = std::dynamic_pointer_cast<Linear>(blocks["caption_projection.linear_2"])->forward(ctx, ctx_emb);
ctx_emb = std::dynamic_pointer_cast<Linear>(blocks["y_embedder.y_proj.fc2"])->forward(ctx, ctx_emb);
for (int i = 0; i < config.num_layers; ++i) {
auto block = std::dynamic_pointer_cast<PixArtBlock>(blocks["transformer_blocks." + std::to_string(i)]);
auto block = std::dynamic_pointer_cast<PixArtBlock>(blocks["blocks." + std::to_string(i)]);
h = block->forward(ctx, h, mod, ctx_emb, context_mask);
sd::ggml_graph_cut::mark_graph_cut(h, "pixart.transformer_blocks." + std::to_string(i), "h");
sd::ggml_graph_cut::mark_graph_cut(h, "pixart.blocks." + std::to_string(i), "h");
}
// scale_shift_table + emb -> (shift, scale) for the affine-free final norm
auto tail_table = params["scale_shift_table"];
auto tail_table = params["final_layer.scale_shift_table"];
if (tail_table->type != GGML_TYPE_F32) {
tail_table = ggml_cast(ctx->ggml_ctx, tail_table, GGML_TYPE_F32);
}
@@ -277,9 +286,9 @@ namespace PixArt {
auto parts = ggml_ext_chunk(ctx->ggml_ctx,
ggml_reshape_2d(ctx->ggml_ctx, ggml_ext_cont(ctx->ggml_ctx, ss), config.hidden_size * 2, N),
2, 0);
h = std::dynamic_pointer_cast<LayerNorm>(blocks["norm_out"])->forward(ctx, h);
h = std::dynamic_pointer_cast<LayerNorm>(blocks["final_layer.norm_final"])->forward(ctx, h);
h = modulate(ctx->ggml_ctx, h, parts[0], parts[1]);
h = std::dynamic_pointer_cast<Linear>(blocks["proj_out"])->forward(ctx, h); // [N, hp*wp, p*p*out_ch]
h = std::dynamic_pointer_cast<Linear>(blocks["final_layer.linear"])->forward(ctx, h); // [N, hp*wp, p*p*out_ch]
h = DiT::unpatchify(ctx->ggml_ctx, h, hp, wp, static_cast<int>(p), static_cast<int>(p), false);
return h; // [N, out_channels, H, W]
}
+7 -5
View File
@@ -140,7 +140,8 @@ namespace ZImage {
int64_t num_kv_heads,
bool qk_norm,
bool norm_elementwise_affine = true,
bool split_qkv = false)
bool split_qkv = false,
float qk_norm_eps = 1e-6f)
: head_dim(head_dim), num_heads(num_heads), num_kv_heads(num_kv_heads), qk_norm(qk_norm), split_qkv(split_qkv) {
float scale = 1.f;
if (split_qkv) {
@@ -153,8 +154,8 @@ namespace ZImage {
blocks["out"] = std::make_shared<Linear>(num_heads * head_dim, hidden_size, false, false, false, scale);
}
if (qk_norm) {
blocks["q_norm"] = std::make_shared<RMSNorm>(head_dim, 1e-06f, norm_elementwise_affine);
blocks["k_norm"] = std::make_shared<RMSNorm>(head_dim, 1e-06f, norm_elementwise_affine);
blocks["q_norm"] = std::make_shared<RMSNorm>(head_dim, qk_norm_eps, norm_elementwise_affine);
blocks["k_norm"] = std::make_shared<RMSNorm>(head_dim, qk_norm_eps, norm_elementwise_affine);
}
}
@@ -318,9 +319,10 @@ namespace ZImage {
bool qk_norm,
bool modulation = true,
bool norm_elementwise_affine = true,
bool split_qkv = false)
bool split_qkv = false,
float qk_norm_eps = 1e-6f)
: modulation(modulation) {
blocks["attention"] = std::make_shared<JointAttention>(hidden_size, head_dim, num_heads, num_kv_heads, qk_norm, norm_elementwise_affine, split_qkv);
blocks["attention"] = std::make_shared<JointAttention>(hidden_size, head_dim, num_heads, num_kv_heads, qk_norm, norm_elementwise_affine, split_qkv, qk_norm_eps);
blocks["feed_forward"] = std::make_shared<FeedForward>(hidden_size, hidden_size, multiple_of, ffn_dim_multiplier);
blocks["attention_norm1"] = std::make_shared<RMSNorm>(hidden_size, norm_eps, norm_elementwise_affine);
blocks["ffn_norm1"] = std::make_shared<RMSNorm>(hidden_size, norm_eps, norm_elementwise_affine);
+102 -33
View File
@@ -50,6 +50,8 @@ namespace LLM {
GEMMA4_12B,
GPT_OSS_20B,
LLADA2_MOE,
BAILING_MOE,
QWEN2,
ARCH_COUNT,
};
@@ -64,6 +66,8 @@ namespace LLM {
"gemma4_12b",
"gpt_oss_20b",
"llada2_moe",
"bailing_moe",
"qwen2",
};
enum class MLPActivation {
@@ -225,7 +229,7 @@ namespace LLM {
config.intermediate_size = 9216;
config.num_layers = 26;
config.vocab_size = 256000;
} else if (arch == LLMArch::LLADA2_MOE) {
} else if (arch == LLMArch::LLADA2_MOE || arch == LLMArch::BAILING_MOE) {
config.head_dim = 128;
config.num_heads = 16;
config.num_kv_heads = 4;
@@ -240,7 +244,7 @@ namespace LLM {
config.max_position_embeddings = 16384;
config.rope_thetas = {600000.f};
config.qkv_fused = true;
config.bidirectional = true;
config.bidirectional = arch == LLMArch::LLADA2_MOE;
config.partial_rotary = 0.5f;
config.num_experts = 256;
config.num_experts_per_tok = 8;
@@ -250,6 +254,18 @@ namespace LLM {
config.n_group = 8;
config.topk_group = 4;
config.routed_scaling_factor = 2.5f;
if (arch == LLMArch::BAILING_MOE) {
config.vocab_size = 157184;
config.max_position_embeddings = 32768;
}
} else if (arch == LLMArch::QWEN2) {
config.hidden_size = 1536;
config.intermediate_size = 8960;
config.num_heads = 12;
config.num_kv_heads = 2;
config.vocab_size = 151936;
config.max_position_embeddings = 32768;
config.bidirectional = true;
} else if (arch == LLMArch::GPT_OSS_20B) {
config.head_dim = 64;
config.num_heads = 64;
@@ -283,22 +299,22 @@ namespace LLM {
if (contains(name, "attn.q_proj")) {
config.llama_cpp_style = true;
}
if (contains(name, "visual.patch_embed.proj.1.weight")) {
if (ends_with(name, "visual.patch_embed.proj.1.weight")) {
config.vision.split_patch_embed = true;
}
if (contains(name, "visual.patch_embed.proj.0.weight")) {
if (ends_with(name, "visual.patch_embed.proj.0.weight")) {
config.vision.patch_size = static_cast<int>(tensor_storage.ne[0]);
config.vision.in_channels = tensor_storage.ne[2];
config.vision.hidden_size = tensor_storage.ne[3];
}
// HF-format checkpoints keep the patch embed unsplit under a single name.
if (contains(name, "visual.patch_embed.proj.weight")) {
if (ends_with(name, "visual.patch_embed.proj.weight")) {
config.vision.patch_size = static_cast<int>(tensor_storage.ne[0]);
}
if (contains(name, "visual.patch_embed.bias") || contains(name, "visual.patch_embed.proj.bias")) {
config.vision.hidden_size = tensor_storage.ne[0];
}
if (contains(name, "visual.pos_embed.weight")) {
if (ends_with(name, "visual.pos_embed.weight") && tensor_storage.n_dims == 2) {
config.vision.hidden_size = tensor_storage.ne[0];
config.vision.num_position_embeddings = static_cast<int>(tensor_storage.ne[1]);
}
@@ -332,7 +348,7 @@ namespace LLM {
}
}
}
if (contains(name, "embed_tokens.weight")) {
if (ends_with(name, "embed_tokens.weight") && tensor_storage.n_dims == 2) {
config.hidden_size = tensor_storage.ne[0];
config.vocab_size = tensor_storage.ne[1];
}
@@ -471,6 +487,8 @@ namespace LLM {
int64_t n_group;
int64_t topk_group;
float routed_scaling_factor;
bool image_router;
bool fused_experts = false;
void init_params(ggml_context* ctx,
const String2TensorStorage& tensor_storage_map = {},
@@ -488,6 +506,10 @@ namespace LLM {
// scores and the group sums match.
params["gate.weight"] = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, hidden_size, num_experts);
params["gate.expert_bias"] = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, num_experts);
if (image_router) {
params["image_gate.weight"] = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, hidden_size, num_experts);
params["image_gate.expert_bias"] = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, num_experts);
}
ggml_type gate_type = supported_type(get_type(prefix + "experts.gate_proj.weight", tensor_storage_map, GGML_TYPE_F32), hidden_size);
ggml_type up_type = supported_type(get_type(prefix + "experts.up_proj.weight", tensor_storage_map, GGML_TYPE_F32), hidden_size);
@@ -506,8 +528,14 @@ namespace LLM {
}
};
declare_experts("experts.gate_proj.weight", gate_type, hidden_size, moe_intermediate_size);
declare_experts("experts.up_proj.weight", up_type, hidden_size, moe_intermediate_size);
fused_experts = tensor_storage_map.count(prefix + "experts.gate_up_proj.weight") != 0;
if (fused_experts) {
auto type = supported_type(get_type(prefix + "experts.gate_up_proj.weight", tensor_storage_map, GGML_TYPE_F32), hidden_size);
declare_experts("experts.gate_up_proj.weight", type, hidden_size, 2 * moe_intermediate_size);
} else {
declare_experts("experts.gate_proj.weight", gate_type, hidden_size, moe_intermediate_size);
declare_experts("experts.up_proj.weight", up_type, hidden_size, moe_intermediate_size);
}
declare_experts("experts.down_proj.weight", down_type, moe_intermediate_size, hidden_size);
}
@@ -519,7 +547,8 @@ namespace LLM {
num_experts_per_tok(config.num_experts_per_tok),
n_group(config.n_group),
topk_group(config.topk_group),
routed_scaling_factor(config.routed_scaling_factor) {
routed_scaling_factor(config.routed_scaling_factor),
image_router(config.arch == LLMArch::BAILING_MOE) {
if (config.num_shared_experts > 0) {
blocks["shared_experts"] = std::make_shared<MLP>(config.hidden_size,
config.moe_intermediate_size * config.num_shared_experts,
@@ -582,7 +611,7 @@ namespace LLM {
return ggml_mul_mat_id(ctx->ggml_ctx, w, x, selected_experts);
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) {
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x, ggml_tensor* image_mask = nullptr) {
// x: [N, n_token, hidden_size]
GGML_ASSERT(num_experts > 0 && num_experts_per_tok > 0);
GGML_ASSERT(n_group > 0 && topk_group > 0 && num_experts % n_group == 0);
@@ -596,10 +625,21 @@ namespace LLM {
auto logits = ggml_mul_mat(gctx, params["gate.weight"], x);
logits = ggml_reshape_2d(gctx, logits, num_experts, n_token_total);
auto scores = ggml_sigmoid(gctx, logits); // [num_experts, tokens]
auto bias = params["gate.expert_bias"];
if (image_router) {
GGML_ASSERT(image_mask != nullptr);
auto mask = ggml_reshape_2d(gctx, image_mask, 1, n_token_total);
auto inverse_mask = ggml_scale_bias(gctx, mask, -1.f, 1.f);
auto image_logits = ggml_reshape_2d(gctx, ggml_mul_mat(gctx, params["image_gate.weight"], x), num_experts, n_token_total);
logits = ggml_add(gctx, ggml_mul(gctx, logits, inverse_mask), ggml_mul(gctx, image_logits, mask));
bias = ggml_add(gctx,
ggml_mul(gctx, ggml_repeat_4d(gctx, bias, num_experts, n_token_total, 1, 1), inverse_mask),
ggml_mul(gctx, ggml_repeat_4d(gctx, params["image_gate.expert_bias"], num_experts, n_token_total, 1, 1), mask));
}
auto scores = ggml_sigmoid(gctx, logits);
// The bias steers selection only; the combine weights come from the unbiased scores.
auto routing = ggml_add(gctx, scores, params["gate.expert_bias"]);
auto routing = ggml_add(gctx, scores, bias);
routing = ggml_add(gctx, routing, group_limited_mask(ctx, routing, n_token_total));
auto selected_experts = ggml_argsort_top_k(gctx, routing, (int)num_experts_per_tok); // [top_k, tokens]
@@ -614,12 +654,17 @@ namespace LLM {
weights = ggml_scale(gctx, weights, routed_scaling_factor);
weights = ggml_reshape_3d(gctx, weights, 1, num_experts_per_tok, n_token_total);
auto xf = ggml_reshape_3d(gctx, x, hidden_size, 1, n_token_total);
auto gate = expert_linear(ctx, "experts.gate_proj.weight", xf, selected_experts);
auto up = expert_linear(ctx, "experts.up_proj.weight", xf, selected_experts);
auto activated = ggml_swiglu_split(gctx, gate, up);
auto experts = expert_linear(ctx, "experts.down_proj.weight", activated, selected_experts);
experts = ggml_mul(gctx, experts, weights);
auto xf = ggml_reshape_3d(gctx, x, hidden_size, 1, n_token_total);
ggml_tensor* activated;
if (fused_experts) {
activated = ggml_swiglu(gctx, expert_linear(ctx, "experts.gate_up_proj.weight", xf, selected_experts));
} else {
auto gate = expert_linear(ctx, "experts.gate_proj.weight", xf, selected_experts);
auto up = expert_linear(ctx, "experts.up_proj.weight", xf, selected_experts);
activated = ggml_swiglu_split(gctx, gate, up);
}
auto experts = expert_linear(ctx, "experts.down_proj.weight", activated, selected_experts);
experts = ggml_mul(gctx, experts, weights);
ggml_tensor* out = nullptr;
for (int64_t i = 0; i < num_experts_per_tok; ++i) {
@@ -1428,7 +1473,9 @@ namespace LLM {
ggml_tensor* x,
ggml_tensor* input_pos,
ggml_tensor* attention_mask = nullptr,
int rope_index = 0) {
int rope_index = 0,
ggml_tensor* rope_cos = nullptr,
ggml_tensor* rope_sin = nullptr) {
// x: [N, n_token, hidden_size]
int64_t n_token = x->ne[1];
int64_t N = x->ne[2];
@@ -1471,13 +1518,28 @@ namespace LLM {
v = ggml_rms_norm(ctx->ggml_ctx, v, rms_norm_eps);
}
if (arch == LLMArch::MISTRAL_SMALL_3_2) {
if (rope_cos != nullptr) {
GGML_ASSERT(rope_sin != nullptr);
// Bailing video RoPE interleaves spatial frequencies within a partial NEOX head.
auto rotate = [&](ggml_tensor* input) {
auto gctx = ctx->ggml_ctx;
int64_t half = rope_cos->ne[0];
auto first = ggml_ext_slice(gctx, input, 0, 0, half);
auto second = ggml_ext_slice(gctx, input, 0, half, 2 * half);
auto left = ggml_sub(gctx, ggml_mul(gctx, first, rope_cos), ggml_mul(gctx, second, rope_sin));
auto right = ggml_add(gctx, ggml_mul(gctx, second, rope_cos), ggml_mul(gctx, first, rope_sin));
auto rotated = ggml_concat(gctx, left, right, 0);
return ggml_concat(gctx, rotated, ggml_ext_slice(gctx, input, 0, 2 * half, input->ne[0]), 0);
};
q = rotate(q);
k = rotate(k);
} else if (arch == LLMArch::MISTRAL_SMALL_3_2) {
q = ggml_rope_ext(ctx->ggml_ctx, q, input_pos, nullptr, 128, GGML_ROPE_TYPE_NORMAL, 8192, 1000000000.f, 1.f, 0.f, 1.f, 32.f, 1.f);
k = ggml_rope_ext(ctx->ggml_ctx, k, input_pos, nullptr, 128, GGML_ROPE_TYPE_NORMAL, 8192, 1000000000.f, 1.f, 0.f, 1.f, 32.f, 1.f);
} else if (arch == LLMArch::MINISTRAL_3_3B) {
q = ggml_rope_ext(ctx->ggml_ctx, q, input_pos, nullptr, 128, GGML_ROPE_TYPE_NEOX, 262144, 1000000.f, 1.f, 0.f, 1.f, 32.f, 1.f);
k = ggml_rope_ext(ctx->ggml_ctx, k, input_pos, nullptr, 128, GGML_ROPE_TYPE_NEOX, 262144, 1000000.f, 1.f, 0.f, 1.f, 32.f, 1.f);
} else if (arch == LLMArch::QWEN3) {
} else if (arch == LLMArch::QWEN3 || arch == LLMArch::QWEN2) {
q = ggml_rope_ext(ctx->ggml_ctx, q, input_pos, nullptr, 128, GGML_ROPE_TYPE_NEOX, 40960, 1000000.f, 1.f, 0.f, 1.f, 32.f, 1.f);
k = ggml_rope_ext(ctx->ggml_ctx, k, input_pos, nullptr, 128, GGML_ROPE_TYPE_NEOX, 40960, 1000000.f, 1.f, 0.f, 1.f, 32.f, 1.f);
} else if (arch == LLMArch::GPT_OSS_20B) {
@@ -1655,7 +1717,7 @@ namespace LLM {
v_attn = ggml_reshape_3d(ctx->ggml_ctx, v_attn, n_token, head_dim, num_kv_heads * N);
auto kq = ggml_mul_mat(ctx->ggml_ctx, k, q);
ggml_mul_mat_set_prec(kq, GGML_PREC_F32);
ggml_prec_set_acc(kq, GGML_PREC_F32);
kq = ggml_scale_inplace(ctx->ggml_ctx, kq, 1.0f / std::sqrt(static_cast<float>(head_dim)));
if (attention_mask != nullptr) {
kq = ggml_add_inplace(ctx->ggml_ctx, kq, attention_mask);
@@ -1724,7 +1786,7 @@ namespace LLM {
blocks["self_attn"] = std::make_shared<Attention>(config, sliding_attention == 0);
if (config.arch == LLMArch::GPT_OSS_20B) {
blocks["mlp"] = std::make_shared<GPTOSSMLP>(config);
} else if (config.arch == LLMArch::LLADA2_MOE && layer_index >= config.first_k_dense_replace) {
} else if ((config.arch == LLMArch::LLADA2_MOE || config.arch == LLMArch::BAILING_MOE) && layer_index >= config.first_k_dense_replace) {
blocks["mlp"] = std::make_shared<LLaDA2MoEMLP>(config);
} else {
blocks["mlp"] = std::make_shared<MLP>(config.hidden_size,
@@ -1746,7 +1808,10 @@ namespace LLM {
ggml_tensor* x,
ggml_tensor* input_pos,
ggml_tensor* attention_mask = nullptr,
ggml_tensor* sliding_attention_mask = nullptr) {
ggml_tensor* sliding_attention_mask = nullptr,
ggml_tensor* image_mask = nullptr,
ggml_tensor* rope_cos = nullptr,
ggml_tensor* rope_sin = nullptr) {
// x: [N, n_token, hidden_size]
auto self_attn = std::dynamic_pointer_cast<Attention>(blocks["self_attn"]);
auto input_layernorm = std::dynamic_pointer_cast<LLMRMSNorm>(blocks["input_layernorm"]);
@@ -1768,7 +1833,7 @@ namespace LLM {
auto residual = x;
x = input_layernorm->forward(ctx, x);
x = self_attn->forward(ctx, x, input_pos, block_attention_mask, rope_index);
x = self_attn->forward(ctx, x, input_pos, block_attention_mask, rope_index, rope_cos, rope_sin);
if (post_attention_norm != nullptr) {
x = post_attention_norm->forward(ctx, x);
}
@@ -1782,7 +1847,7 @@ namespace LLM {
} else if (auto moe_mlp = std::dynamic_pointer_cast<LLaDA2MoEMLP>(blocks["mlp"])) {
// LLaDA2 is dense for the first first_k_dense_replace layers and MoE afterwards,
// so the block type varies per layer rather than per arch.
x = moe_mlp->forward(ctx, x);
x = moe_mlp->forward(ctx, x, image_mask);
} else {
auto mlp = std::dynamic_pointer_cast<MLP>(blocks["mlp"]);
x = mlp->forward(ctx, x);
@@ -1804,10 +1869,11 @@ namespace LLM {
protected:
int64_t num_layers;
LLMConfig config;
std::string graph_cut_prefix;
public:
TextModel(const LLMConfig& config)
: num_layers(config.num_layers), config(config) {
TextModel(const LLMConfig& config, const std::string& graph_cut_prefix = "llm.text")
: num_layers(config.num_layers), config(config), graph_cut_prefix(graph_cut_prefix) {
blocks["embed_tokens"] = std::shared_ptr<GGMLBlock>(new Embedding(config.vocab_size, config.hidden_size));
for (int i = 0; i < num_layers; i++) {
blocks["layers." + std::to_string(i)] = std::shared_ptr<GGMLBlock>(new TransformerBlock(config, i));
@@ -1831,7 +1897,10 @@ namespace LLM {
std::set<int> out_layers,
const std::vector<std::vector<std::pair<int, ggml_tensor*>>>& deepstack_image_embeds = {},
ggml_tensor* sliding_attention_mask = nullptr,
bool return_all_hidden_states = false) {
bool return_all_hidden_states = false,
ggml_tensor* image_mask = nullptr,
ggml_tensor* rope_cos = nullptr,
ggml_tensor* rope_sin = nullptr) {
auto norm = config.final_norm ? std::dynamic_pointer_cast<LLMRMSNorm>(blocks["norm"])
: nullptr;
std::vector<ggml_tensor*> intermediate_outputs;
@@ -1843,18 +1912,18 @@ namespace LLM {
intermediate_outputs.push_back(x);
}
sd::ggml_graph_cut::mark_graph_cut(x, "llm.text.prelude", "x");
sd::ggml_graph_cut::mark_graph_cut(x, graph_cut_prefix + ".prelude", "x");
for (int i = 0; i < num_layers; i++) {
auto block = std::dynamic_pointer_cast<TransformerBlock>(blocks["layers." + std::to_string(i)]);
x = block->forward(ctx, x, input_pos, attention_mask, sliding_attention_mask);
x = block->forward(ctx, x, input_pos, attention_mask, sliding_attention_mask, image_mask, rope_cos, rope_sin);
if (i < static_cast<int>(deepstack_image_embeds.size())) {
x = add_deepstack_image_embeds(ctx, x, deepstack_image_embeds[static_cast<size_t>(i)]);
}
if (return_all_hidden_states || out_layers.size() > 1) {
x = ggml_cont(ctx->ggml_ctx, x);
}
sd::ggml_graph_cut::mark_graph_cut(x, "llm.text.layers." + std::to_string(i), "x");
sd::ggml_graph_cut::mark_graph_cut(x, graph_cut_prefix + ".layers." + std::to_string(i), "x");
if (return_all_hidden_states) {
if (i + 1 < num_layers) {
intermediate_outputs.push_back(x);
+159
View File
@@ -0,0 +1,159 @@
#ifndef __SD_MODEL_TE_MING_IMAGE_TE_HPP__
#define __SD_MODEL_TE_MING_IMAGE_TE_HPP__
#include "llm.hpp"
namespace MingImageTE {
struct MingImageTEConfig {
LLM::LLMConfig backbone;
LLM::LLMConfig connector;
int64_t num_queries = 256;
int64_t caption_dim = 2560;
int64_t diffusion_dim = 3840;
static MingImageTEConfig detect_from_weights(const String2TensorStorage& tensors, const std::string& prefix) {
MingImageTEConfig config;
for (const auto& entry : tensors) {
if (starts_with(entry.first, prefix + ".backbone.") &&
contains(entry.first, ".mlp.experts.") && entry.second.type == GGML_TYPE_I8) {
throw std::runtime_error("Ming-Image INT8/W4A8 text encoder experts are not supported; use the BF16 text encoder");
}
}
bool vision = false;
config.backbone = LLM::LLMConfig::detect_from_weights(tensors, prefix + ".backbone.", LLM::LLMArch::BAILING_MOE, vision);
config.connector = LLM::LLMConfig::detect_from_weights(tensors, prefix + ".connector.", LLM::LLMArch::QWEN2, vision);
const auto query = tensors.find(prefix + ".query_tokens_dict.16x16");
const auto projection = tensors.find(prefix + ".proj_out.weight");
const auto direct = tensors.find(prefix + ".proj_directvlm.1.weight");
if (query == tensors.end() || projection == tensors.end() || direct == tensors.end()) {
throw std::runtime_error("Ming-Image requires the learned queries, connector and both condition projections");
}
config.num_queries = query->second.ne[1];
config.caption_dim = projection->second.ne[1];
config.diffusion_dim = direct->second.ne[1];
if (config.num_queries != 256 || config.caption_dim != 2560 || config.diffusion_dim != 3840 ||
config.backbone.num_layers != 20 || config.backbone.hidden_size != 2048 ||
config.connector.num_layers != 28 || config.connector.hidden_size != 1536) {
throw std::runtime_error("unsupported Ming-Image text encoder configuration");
}
LOG_VERBOSE("ming_image_te: queries = %" PRId64 ", caption_dim = %" PRId64 ", diffusion_dim = %" PRId64,
config.num_queries, config.caption_dim, config.diffusion_dim);
return config;
}
};
struct ConnectorModel : LLM::TextModel {
explicit ConnectorModel(const LLM::LLMConfig& config)
: LLM::TextModel(config, "ming_image.connector") {
blocks.erase("embed_tokens");
}
};
class MingImageTextModel : public GGMLBlock {
MingImageTEConfig config;
void init_params(ggml_context* ctx, const String2TensorStorage& tensors = {}, const std::string prefix = "") override {
params["query_tokens_dict.16x16"] = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, config.backbone.hidden_size, config.num_queries);
}
public:
explicit MingImageTextModel(const MingImageTEConfig& config)
: config(config) {
blocks["backbone"] = std::make_shared<LLM::TextModel>(config.backbone, "ming_image.backbone");
blocks["connector"] = std::make_shared<ConnectorModel>(config.connector);
blocks["proj_in"] = std::make_shared<Linear>(config.backbone.hidden_size, config.connector.hidden_size);
blocks["proj_out"] = std::make_shared<Linear>(config.connector.hidden_size, config.caption_dim);
blocks["proj_directvlm.0"] = std::make_shared<RMSNorm>(config.backbone.hidden_size * 3, 1e-5f);
blocks["proj_directvlm.1"] = std::make_shared<Linear>(config.backbone.hidden_size * 3, config.diffusion_dim);
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* ids, int64_t prompt_length, ggml_tensor* mask, ggml_tensor* image_mask, ggml_tensor* cos, ggml_tensor* sin, ggml_tensor* connector_positions) {
auto gctx = ctx->ggml_ctx;
auto backbone = std::dynamic_pointer_cast<LLM::TextModel>(blocks["backbone"]);
auto connector = std::dynamic_pointer_cast<ConnectorModel>(blocks["connector"]);
auto x = backbone->embed(ctx, ids);
const int64_t query_start = prompt_length + 1;
auto before = ggml_ext_slice(gctx, x, 1, 0, query_start);
auto after = ggml_ext_slice(gctx, x, 1, query_start + config.num_queries, x->ne[1]);
x = ggml_concat(gctx, ggml_concat(gctx, before, params["query_tokens_dict.16x16"], 1), after, 1);
// HF hidden_states[20] includes final RMSNorm; sd.cpp selects it as num_layers + 1.
x = backbone->forward_embeds(ctx, x, nullptr, mask, {5, 12, 21}, {}, nullptr, false, image_mask, cos, sin);
auto direct = ggml_ext_slice(gctx, x, 1, 0, prompt_length);
direct = std::dynamic_pointer_cast<RMSNorm>(blocks["proj_directvlm.0"])->forward(ctx, direct);
direct = std::dynamic_pointer_cast<Linear>(blocks["proj_directvlm.1"])->forward(ctx, direct);
auto queries = ggml_ext_slice(gctx, x, 0, config.backbone.hidden_size * 2, config.backbone.hidden_size * 3);
queries = ggml_ext_slice(gctx, queries, 1, query_start, query_start + config.num_queries);
queries = std::dynamic_pointer_cast<Linear>(blocks["proj_in"])->forward(ctx, queries);
queries = connector->forward_embeds(ctx, queries, connector_positions, nullptr, {});
queries = std::dynamic_pointer_cast<Linear>(blocks["proj_out"])->forward(ctx, queries);
queries = ggml_pad(gctx, queries, static_cast<int>(config.diffusion_dim - config.caption_dim), 0, 0, 0);
return ggml_concat(gctx, queries, direct, 1);
}
};
struct MingImageTextRunner : GGMLRunner {
MingImageTEConfig config;
MingImageTextModel model;
MingImageTextRunner(ggml_backend_t backend, const String2TensorStorage& tensors, const std::string& prefix, std::shared_ptr<RunnerWeightManager> weight_manager = nullptr)
: GGMLRunner(backend, weight_manager), config(MingImageTEConfig::detect_from_weights(tensors, prefix)), model(config) {
model.init(params_ctx, tensors, prefix);
}
std::string get_desc() override { return "ming_image_text"; }
void get_param_tensors(std::map<std::string, ggml_tensor*>& tensors, const std::string& prefix) {
model.get_param_tensors(tensors, prefix);
}
void get_param_tensor_ops(std::map<ggml_tensor*, enum ggml_op>& ops) {
model.get_param_tensor_ops(ops);
}
sd::Tensor<float> compute(int n_threads, const std::vector<int>& tokens) {
const int64_t prompt_length = tokens.size();
const int64_t query_start = prompt_length + 1;
const int64_t total = prompt_length + config.num_queries + 2;
std::vector<int32_t> ids(tokens.begin(), tokens.end());
ids.push_back(157158);
ids.insert(ids.end(), config.num_queries, 157157);
ids.push_back(157159);
auto input = sd::Tensor<int32_t>({total}, std::move(ids));
sd::Tensor<float> attention_mask({total, total});
sd::Tensor<float> image_mask({1, total});
sd::Tensor<float> cos({32, 1, total}), sin({32, 1, total});
std::vector<int32_t> positions(config.num_queries);
std::iota(positions.begin(), positions.end(), 0);
auto connector_positions = sd::Tensor<int32_t>({config.num_queries}, std::move(positions));
for (int64_t token = 0; token < total; ++token) {
const bool query = token >= query_start && token < query_start + config.num_queries;
image_mask[token] = query ? 1.f : 0.f;
for (int64_t key = 0; key < total; ++key) {
attention_mask[key + token * total] = key > token ? -INFINITY : 0.f;
}
// A 16x16 query bank is represented upstream as a [1, 2, 512] image grid,
// then spatially merged and centered to [1, 1, 256].
int64_t temporal = query ? query_start : (token == total - 1 ? query_start + 1 : token);
int64_t width = query ? token - 127 : temporal;
for (int j = 0; j < 32; ++j) {
int64_t position = query && j < 24 && j % 2 ? width : temporal;
float frequency = 1.f / std::pow(600000.f, static_cast<float>(2 * j) / 64.f);
float angle = static_cast<float>(position) * frequency;
cos[token * 32 + j] = std::cos(angle);
sin[token * 32 + j] = std::sin(angle);
}
}
auto graph = [&]() {
auto gf = new_graph_custom(LLM::LLM_GRAPH_SIZE);
auto ctx = get_context();
auto out = model.forward(&ctx, make_input(input), prompt_length, make_input(attention_mask),
make_input(image_mask), make_input(cos), make_input(sin), make_input(connector_positions));
ggml_build_forward_expand(gf, out);
return gf;
};
return take_or_empty(GGMLRunner::compute(graph, n_threads));
}
};
}
#endif // __SD_MODEL_TE_MING_IMAGE_TE_HPP__
+7 -1
View File
@@ -1084,7 +1084,7 @@ namespace WAN {
_conv_num = 34;
_enc_conv_num = 26;
} else if (version == VERSION_QWEN_IMAGE_LAYERED) {
} else if (version == VERSION_QWEN_IMAGE_LAYERED || version == VERSION_MING_IMAGE) {
input_channels = 4;
}
@@ -1423,11 +1423,17 @@ namespace WAN {
}
sd::Tensor<float> diffusion_to_vae_latents(const sd::Tensor<float>& latents) override {
if (version == VERSION_MING_IMAGE) {
return latents / 8.0064f;
}
auto [mean_tensor, std_tensor] = get_latents_mean_std(latents);
return (latents * std_tensor) / scale_factor + mean_tensor;
}
sd::Tensor<float> vae_to_diffusion_latents(const sd::Tensor<float>& latents) override {
if (version == VERSION_MING_IMAGE) {
return latents * 8.0064f;
}
auto [mean_tensor, std_tensor] = get_latents_mean_std(latents);
return ((latents - mean_tensor) * scale_factor) / std_tensor;
}
+7 -3
View File
@@ -47,10 +47,14 @@ bool read_gguf_file(const std::string& file_path,
gguf_context* ctx_gguf_ = nullptr;
ggml_context* ctx_meta_ = nullptr;
ctx_gguf_ = gguf_init_from_file(file_path.c_str(), {true, &ctx_meta_});
GGUFReader gguf_reader;
bool probe_ok = gguf_reader.load(file_path);
if (!probe_ok || !gguf_reader.has_tensors_beyond_ggml_limits()) {
ctx_gguf_ = gguf_init_from_file(file_path.c_str(), {true, &ctx_meta_});
}
if (!ctx_gguf_) {
GGUFReader gguf_reader;
if (!gguf_reader.load(file_path)) {
if (!probe_ok && !gguf_reader.load(file_path)) {
set_error(error, "failed to open '" + file_path + "' with GGUFReader");
return false;
}
+71 -51
View File
@@ -35,23 +35,72 @@ enum class GGUFMetadataType : uint32_t {
class GGUFReader {
private:
std::vector<GGUFTensorInfo> tensors_;
bool has_wide_tensors_ = false;
uint64_t remaining_bytes_ = 0;
size_t data_offset_;
size_t alignment_ = 32; // default alignment is 32
template <typename T>
bool safe_read(std::ifstream& fin, T& value) {
fin.read(reinterpret_cast<char*>(&value), sizeof(T));
return fin.good();
return safe_read(fin, reinterpret_cast<char*>(&value), sizeof(T));
}
bool safe_read(std::ifstream& fin, char* buffer, size_t size) {
if (size > remaining_bytes_)
return false;
fin.read(buffer, size);
return fin.good();
if (!fin.good())
return false;
remaining_bytes_ -= size;
return true;
}
bool safe_seek(std::ifstream& fin, std::streamoff offset, std::ios::seekdir dir) {
fin.seekg(offset, dir);
return fin.good();
bool safe_skip(std::ifstream& fin, uint64_t count, uint64_t element_size = 1) {
if (count > remaining_bytes_ / element_size)
return false;
uint64_t size = count * element_size;
fin.seekg(static_cast<std::streamoff>(size), std::ios::cur);
if (!fin.good())
return false;
remaining_bytes_ -= size;
return true;
}
bool skip_metadata_values(std::ifstream& fin, GGUFMetadataType type, uint64_t count) {
switch (type) {
case GGUFMetadataType::UINT8:
case GGUFMetadataType::INT8:
case GGUFMetadataType::BOOL:
return safe_skip(fin, count);
case GGUFMetadataType::UINT16:
case GGUFMetadataType::INT16:
return safe_skip(fin, count, 2);
case GGUFMetadataType::UINT32:
case GGUFMetadataType::INT32:
case GGUFMetadataType::FLOAT32:
return safe_skip(fin, count, 4);
case GGUFMetadataType::UINT64:
case GGUFMetadataType::INT64:
case GGUFMetadataType::FLOAT64:
return safe_skip(fin, count, 8);
case GGUFMetadataType::STRING:
if (count > remaining_bytes_ / sizeof(uint64_t))
return false;
for (uint64_t i = 0; i < count; i++) {
uint64_t len = 0;
if (!safe_read(fin, len) || !safe_skip(fin, len))
return false;
}
return true;
default:
LOG_ERROR("Unknown metadata type=%u", static_cast<uint32_t>(type));
return false;
}
}
bool read_metadata(std::ifstream& fin) {
@@ -84,52 +133,12 @@ private:
return true;
}
switch (static_cast<GGUFMetadataType>(type)) {
case GGUFMetadataType::UINT8:
case GGUFMetadataType::INT8:
case GGUFMetadataType::BOOL:
return safe_seek(fin, 1, std::ios::cur);
case GGUFMetadataType::UINT16:
case GGUFMetadataType::INT16:
return safe_seek(fin, 2, std::ios::cur);
case GGUFMetadataType::UINT32:
case GGUFMetadataType::INT32:
case GGUFMetadataType::FLOAT32:
return safe_seek(fin, 4, std::ios::cur);
case GGUFMetadataType::UINT64:
case GGUFMetadataType::INT64:
case GGUFMetadataType::FLOAT64:
return safe_seek(fin, 8, std::ios::cur);
case GGUFMetadataType::STRING: {
uint64_t len = 0;
if (!safe_read(fin, len))
return false;
return safe_seek(fin, len, std::ios::cur);
}
case GGUFMetadataType::ARRAY: {
uint32_t elem_type = 0;
uint64_t len = 0;
if (!safe_read(fin, elem_type))
return false;
if (!safe_read(fin, len))
return false;
for (uint64_t i = 0; i < len; i++) {
if (!read_metadata(fin))
return false;
}
return true;
}
default:
LOG_ERROR("Unknown metadata type=%u", type);
uint64_t count = 1;
if (type == static_cast<uint32_t>(GGUFMetadataType::ARRAY)) {
if (!safe_read(fin, type) || !safe_read(fin, count))
return false;
}
return skip_metadata_values(fin, static_cast<GGUFMetadataType>(type), count);
}
GGUFTensorInfo read_tensor_info(std::ifstream& fin) {
@@ -154,6 +163,7 @@ private:
}
if (n_dims > GGML_MAX_DIMS) {
has_wide_tensors_ = true;
for (uint32_t i = GGML_MAX_DIMS; i < n_dims; i++) {
info.shape[GGML_MAX_DIMS - 1] *= info.shape[i]; // stack to last dim;
}
@@ -174,12 +184,20 @@ private:
public:
bool load(const std::string& file_path) {
std::ifstream fin(file_path, std::ios::binary);
std::ifstream fin(file_path, std::ios::binary | std::ios::ate);
if (!fin) {
LOG_ERROR("failed to open '%s'", file_path.c_str());
return false;
}
std::streamoff file_size = fin.tellg();
if (file_size < 0)
return false;
remaining_bytes_ = static_cast<uint64_t>(file_size);
fin.seekg(0, std::ios::beg);
if (!fin.good())
return false;
// --- Header ---
char magic[4];
if (!safe_read(fin, magic, 4) || strncmp(magic, "GGUF", 4) != 0) {
@@ -228,6 +246,8 @@ public:
}
const std::vector<GGUFTensorInfo>& tensors() const { return tensors_; }
bool has_tensors_beyond_ggml_limits() const { return has_wide_tensors_; }
size_t data_offset() const { return data_offset_; }
};
+18 -6
View File
@@ -524,6 +524,9 @@ SDVersion ModelLoader::get_sd_version() const {
return VERSION_LLADA_IMAGE;
}
if (tensor_storage.name.find("model.diffusion_model.cap_embedder.0.weight") != std::string::npos) {
if (tensor_storage_map.find("text_encoders.llm.connector.layers.0.self_attn.q_proj.weight") != tensor_storage_map.end()) {
return VERSION_MING_IMAGE;
}
return VERSION_Z_IMAGE;
}
if (tensor_storage.name.find("double_stream_layers.0.img_instruct_attn.processor.img_to_q.weight") != std::string::npos) {
@@ -532,9 +535,16 @@ SDVersion ModelLoader::get_sd_version() const {
if (tensor_storage.name.find("model.diffusion_model.layers.0.adaLN_sa_ln.weight") != std::string::npos) {
return VERSION_ERNIE_IMAGE;
}
if (tensor_storage.name.find("model.diffusion_model.t_block.1.weight") != std::string::npos &&
tensor_storage_map.find("model.diffusion_model.x_embedder.proj.weight") != tensor_storage_map.end() &&
tensor_storage_map.find("model.diffusion_model.audio_patchify_proj.weight") == tensor_storage_map.end()) {
return VERSION_PIXART;
}
if (tensor_storage.name.find("model.diffusion_model.adaln_single.emb.timestep_embedder.linear_1.bias") != std::string::npos) {
// PixArt shares this signature with LTX-AV; pos_embed.proj is PixArt-only.
if (tensor_storage_map.find("model.diffusion_model.pos_embed.proj.weight") != tensor_storage_map.end()) {
// PixArt shares this timestep embedding with LTX-AV.
if (tensor_storage_map.find("model.diffusion_model.pos_embed.proj.weight") != tensor_storage_map.end() &&
tensor_storage_map.find("model.diffusion_model.adaln_single.linear.weight") != tensor_storage_map.end() &&
tensor_storage_map.find("model.diffusion_model.audio_patchify_proj.weight") == tensor_storage_map.end()) {
return VERSION_PIXART;
}
return VERSION_LTXAV;
@@ -1094,10 +1104,12 @@ bool ModelLoader::load_tensors(on_new_tensor_cb_t on_new_tensor_cb,
if (tensors_to_process.empty()) {
continue;
}
LOG_VERBOSE("loading %zu/%zu tensors from %s",
tensors_to_process.size(),
file_tensors.size(),
file_path.c_str());
if (log_progress) {
LOG_VERBOSE("loading %zu/%zu tensors from %s",
tensors_to_process.size(),
file_tensors.size(),
file_path.c_str());
}
bool is_zip = fdata.is_zip;
+103 -15
View File
@@ -79,6 +79,10 @@ static bool device_supports_param_op(ggml_backend_dev_t device,
if (op == GGML_OP_GET_ROWS) {
ggml_tensor* indices = ggml_new_tensor_1d(ctx, GGML_TYPE_I32, 1);
op_tensor = ggml_get_rows(ctx, weight, indices);
} else if (op == GGML_OP_CUSTOM) {
op_tensor = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, 1);
op_tensor->op = op;
op_tensor->src[0] = weight;
}
if (op_tensor == nullptr) {
ggml_free(ctx);
@@ -246,17 +250,13 @@ bool ModelManager::register_param_tensors(ModelComponent component,
}
ggml_set_name(tensor, name.c_str());
auto state = std::make_unique<TensorState>();
state->name = name;
state->tensor = tensor;
state->component = component;
state->source_file = source_file;
state->source_version = source_version;
auto source = sources.find(name);
if (source != sources.end()) {
state->source = source->second;
state->has_source = true;
}
auto state = std::make_unique<TensorState>();
state->name = name;
state->tensor = tensor;
state->component = component;
state->source_file = source_file;
state->source_version = source_version;
state->sources = find_tensor_sources(*state, sources);
state->residency_mode = residency_mode;
state->compute_backend = compute_backend;
state->params_backend = params_backend;
@@ -511,7 +511,9 @@ bool ModelManager::stage_tensors_to_compute_backend(const std::vector<TensorStat
LOG_ERROR("model manager params backend is null for tensor '%s'", state->name.c_str());
return false;
}
if (state->compute_backend == state->params_backend || state->staged_to_compute_backend) {
// Custom CPU operators must retain host weights even when the runner uses a GPU.
if (state->usage_op == GGML_OP_CUSTOM ||
state->compute_backend == state->params_backend || state->staged_to_compute_backend) {
continue;
}
if (!state->loaded_to_params_backend || state->tensor == nullptr || state->tensor->data == nullptr) {
@@ -757,12 +759,33 @@ bool ModelManager::validate_tensor(const TensorState& state) const {
return true;
}
if (!state.has_source) {
if (state.sources.empty()) {
LOG_ERROR("%s tensor '%s' not in model metadata", model_component_name(state.component), state.name.c_str());
return false;
}
const TensorStorage& tensor_storage = state.source;
TensorStorage tensor_storage = state.sources.front();
if (state.sources.size() > 1) {
const int dim = tensor_storage.n_dims - 1;
if (dim < 0 || dim >= GGML_MAX_DIMS) {
return false;
}
tensor_storage.ne[dim] = 0;
for (const auto& part : state.sources) {
if (part.n_dims != tensor_storage.n_dims || part.ne[dim] < 0 ||
part.ne[dim] > state.tensor->ne[dim] - tensor_storage.ne[dim]) {
LOG_ERROR("invalid tensor part '%s' for '%s'", part.name.c_str(), state.name.c_str());
return false;
}
for (int i = 0; i < GGML_MAX_DIMS; ++i) {
if (i != dim && part.ne[i] != tensor_storage.ne[i]) {
LOG_ERROR("incompatible tensor part '%s' for '%s'", part.name.c_str(), state.name.c_str());
return false;
}
}
tensor_storage.ne[dim] += part.ne[dim];
}
}
if (state.tensor->ne[0] != tensor_storage.ne[0] ||
state.tensor->ne[1] != tensor_storage.ne[1] ||
state.tensor->ne[2] != tensor_storage.ne[2] ||
@@ -831,7 +854,7 @@ bool ModelManager::mmap_params(const std::vector<TensorState*>& states,
}
bool ModelManager::can_mmap_storage(const TensorState& state) const {
if (state.source_file != 0 || !enable_mmap_ || state.residency_mode != ResidencyMode::ParamBackend) {
if (state.sources.size() > 1 || state.source_file != 0 || !enable_mmap_ || state.residency_mode != ResidencyMode::ParamBackend) {
return false;
}
if (state.compute_backend == nullptr || state.params_backend == nullptr) {
@@ -941,6 +964,62 @@ bool ModelManager::alloc_params_buffers(const std::vector<TensorState*>& states,
return true;
}
bool ModelManager::load_tensor_parts(TensorState& state) {
auto ctx = std::unique_ptr<ggml_context, decltype(&ggml_free)>(
ggml_init({state.sources.size() * ggml_tensor_overhead(), nullptr, true}), ggml_free);
if (!ctx) {
return false;
}
const size_t size = ggml_nbytes(state.tensor);
std::vector<uint8_t> buffer;
void* data = state.tensor->data;
if (!ggml_backend_buffer_is_host(state.tensor->buffer)) {
buffer.resize(size);
data = buffer.data();
}
std::map<std::string, ggml_tensor*> parts;
std::set<std::string> names;
size_t offset = 0;
for (const auto& source : state.sources) {
auto part = ggml_new_tensor(ctx.get(), state.tensor->type, source.n_dims, source.ne);
const size_t part_size = ggml_nbytes(part);
if (part_size > size - offset) {
return false;
}
part->data = static_cast<uint8_t*>(data) + offset;
parts[source.name] = part;
names.insert(source.name);
offset += part_size;
}
if (offset != size) {
return false;
}
std::set<std::string> loaded;
std::mutex mutex;
auto callback = [&](const TensorStorage& source, ggml_tensor** dst) {
*dst = nullptr;
auto part = parts.find(source.name);
if (part != parts.end()) {
*dst = part->second;
std::lock_guard<std::mutex> lock(mutex);
loaded.insert(source.name);
}
return true;
};
const bool success = state.source_file == 0
? model_loader_.load_tensors(callback, enable_mmap_, &names, false)
: model_loader_.load_file_tensors(state.source_file, state.source_version, callback, names, enable_mmap_);
if (!success || loaded != names) {
return false;
}
if (!buffer.empty()) {
// Upload the assembled tensor once, including for row-split backend buffers.
ggml_backend_tensor_set(state.tensor, buffer.data(), 0, size);
}
state.loaded_to_params_backend = true;
return true;
}
bool ModelManager::load_tensors(const std::vector<TensorState*>& states) {
using ReadGroup = std::pair<ModelLoader::FileId, SDVersion>;
using ReadBatch = std::map<std::string, std::vector<TensorState*>>;
@@ -948,6 +1027,12 @@ bool ModelManager::load_tensors(const std::vector<TensorState*>& states) {
for (auto* state : states) {
if (state == nullptr)
continue;
if (state->sources.size() > 1) {
if (!load_tensor_parts(*state)) {
return false;
}
continue;
}
auto& batches = groups[{state->source_file, state->source_version}];
// The loader supplies one destination per name; only conflicting types need another batch.
auto batch = std::find_if(batches.begin(), batches.end(), [&](const ReadBatch& candidate) {
@@ -1323,6 +1408,9 @@ size_t ModelManager::compute_backend_alloc_size(const std::vector<TensorState*>&
if (state == nullptr || state->tensor == nullptr) {
continue;
}
if (state->usage_op == GGML_OP_CUSTOM && !sd_backend_is_cpu(state->compute_backend)) {
continue;
}
const bool compute_resident =
state->compute_backend == state->params_backend
? state->loaded_to_params_backend
+3 -2
View File
@@ -38,8 +38,7 @@ private:
std::string name;
ggml_tensor* tensor = nullptr;
ModelComponent component = ModelComponent::Count;
TensorStorage source;
bool has_source = false;
std::vector<TensorStorage> sources;
ModelLoader::FileId source_file = 0;
SDVersion source_version = VERSION_COUNT;
@@ -138,6 +137,8 @@ private:
bool apply_loras_to_params(const std::vector<TensorState*>& states);
bool mmap_params(const std::vector<TensorState*>& states,
std::vector<ParamsStorageBlock*>& created_storage_blocks);
static std::vector<TensorStorage> find_tensor_sources(const TensorState& state, const String2TensorStorage& sources);
bool load_tensor_parts(TensorState& state);
bool can_mmap_storage(const TensorState& state) const;
bool alloc_params_buffers(const std::vector<TensorState*>& states,
std::vector<ParamsStorageBlock*>& created_storage_blocks);
+26 -11
View File
@@ -15,6 +15,26 @@ static bool same_tensor_source(const TensorStorage& a, const TensorStorage& b) {
a.int8_convrot_group_size == b.int8_convrot_group_size;
}
std::vector<TensorStorage> ModelManager::find_tensor_sources(const TensorState& state, const String2TensorStorage& sources) {
auto first = sources.find(state.name);
if (first == sources.end()) {
return {};
}
std::vector<TensorStorage> result{first->second};
if (state.component == ModelComponent::LoRA ||
std::equal(first->second.ne, first->second.ne + GGML_MAX_DIMS, state.tensor->ne)) {
return result;
}
for (size_t i = 1;; ++i) {
auto part = sources.find(state.name + "." + std::to_string(i));
if (part == sources.end()) {
break;
}
result.push_back(part->second);
}
return result;
}
void ModelManager::invalidate_sources(const std::unordered_set<TensorState*>& states) {
auto affected = states;
for (const auto& block : params_storage_blocks_) {
@@ -76,20 +96,16 @@ bool ModelManager::set_loader(ModelLoader loader) {
}
std::unordered_set<TensorState*> changed;
for (const auto& state : tensor_states_) {
const auto& sources = sources_for(*state);
auto source = sources.find(state->name);
const bool found = source != sources.end();
if (found != state->has_source || (found && !same_tensor_source(state->source, source->second)) ||
const auto sources = find_tensor_sources(*state, sources_for(*state));
if (sources.size() != state->sources.size() ||
!std::equal(sources.begin(), sources.end(), state->sources.begin(), same_tensor_source) ||
(lora_changed && state->component != ModelComponent::LoRA && state->applied_lora_epoch != UINT64_MAX)) {
changed.insert(state.get());
}
}
invalidate_sources(changed);
for (auto* state : changed) {
const auto& sources = sources_for(*state);
auto source = sources.find(state->name);
state->has_source = source != sources.end();
state->source = state->has_source ? source->second : TensorStorage{};
state->sources = find_tensor_sources(*state, sources_for(*state));
}
if (lora_changed) {
++current_lora_epoch_;
@@ -134,9 +150,8 @@ ModelLoader::FileVersions ModelManager::source_versions(const std::set<ModelComp
versions[state->source_file] = loader.file_revision(state->source_file);
continue;
}
auto source = sources.find(state->name);
if (source != sources.end()) {
versions[source->second.file_id] = source->second.file_revision;
for (const auto& source : find_tensor_sources(*state, sources)) {
versions[source.file_id] = source.file_revision;
}
}
return versions;
+2 -1
View File
@@ -210,7 +210,7 @@ WeightPrefetchResult ModelManager::prefetch_params(
is_optional_missing_tensor(state->name)) {
continue;
}
if (state->compute_backend == state->params_backend) {
if (state->usage_op == GGML_OP_CUSTOM || state->compute_backend == state->params_backend) {
needs_synchronous_load = needs_synchronous_load ||
!state->loaded_to_params_backend;
continue;
@@ -275,6 +275,7 @@ bool ModelManager::activate_prefetched_params(
[&](TensorState* state) {
return state == nullptr || should_ignore(*state) ||
is_optional_missing_tensor(state->name) ||
state->usage_op == GGML_OP_CUSTOM ||
state->compute_backend == state->params_backend ||
state->staged_to_compute_backend;
});
+97 -2
View File
@@ -105,7 +105,25 @@ std::string convert_open_clip_to_hf_clip_name(std::string name) {
std::string convert_llada2_moe_te_name(std::string name);
std::string convert_cond_stage_model_name(std::string name, std::string prefix) {
static std::string convert_ming_image_te_name(std::string name) {
if (name == "llm.query_tokens") {
name = "llm.query_tokens_dict.16x16";
}
static const std::vector<std::pair<std::string, std::string>> name_map = {
{"thinker.", "backbone."},
{"attention.", "self_attn."},
{"self_attn.dense.", "self_attn.o_proj."},
{"gate.proj.", "gate."},
};
replace_with_name_map(name, name_map);
return name;
}
std::string convert_cond_stage_model_name(std::string name, std::string prefix, SDVersion version) {
if (version == VERSION_MING_IMAGE && prefix == "text_encoders." && starts_with(name, "llm.")) {
name = convert_ming_image_te_name(name);
}
static const std::vector<std::pair<std::string, std::string>> clip_name_map{
{"transformer.text_projection.weight", "transformer.text_model.text_projection"},
{"model.text_projection.weight", "transformer.text_model.text_projection"},
@@ -932,6 +950,79 @@ static bool is_diffusers_controlnet_name(const std::string& name) {
return false;
}
static std::string convert_diffusers_dit_to_original_pixart(std::string name) {
static const std::vector<std::pair<std::string, std::string>> prefix_map = {
{"pos_embed.proj.", "x_embedder.proj."},
{"adaln_single.emb.timestep_embedder.linear_1.", "t_embedder.mlp.0."},
{"adaln_single.emb.timestep_embedder.linear_2.", "t_embedder.mlp.2."},
{"adaln_single.emb.resolution_embedder.linear_1.", "csize_embedder.mlp.0."},
{"adaln_single.emb.resolution_embedder.linear_2.", "csize_embedder.mlp.2."},
{"adaln_single.emb.aspect_ratio_embedder.linear_1.", "ar_embedder.mlp.0."},
{"adaln_single.emb.aspect_ratio_embedder.linear_2.", "ar_embedder.mlp.2."},
{"adaln_single.linear.", "t_block.1."},
{"caption_projection.linear_1.", "y_embedder.y_proj.fc1."},
{"caption_projection.linear_2.", "y_embedder.y_proj.fc2."},
{"proj_out.", "final_layer.linear."},
};
for (const auto& entry : prefix_map) {
if (starts_with(name, entry.first)) {
return entry.second + name.substr(entry.first.size());
}
}
if (name == "scale_shift_table") {
return "final_layer.scale_shift_table";
}
const std::string block_prefix = "transformer_blocks.";
if (!starts_with(name, block_prefix)) {
return name;
}
const size_t block_end = name.find('.', block_prefix.size());
if (block_end == std::string::npos) {
return name;
}
const std::string prefix = "blocks." + name.substr(block_prefix.size(), block_end - block_prefix.size()) + ".";
name = name.substr(block_end + 1);
static const std::vector<std::pair<std::string, std::string>> block_map = {
{"attn1.to_q.", "attn.qkv."},
{"attn1.to_out.0.", "attn.proj."},
{"attn2.to_q.", "cross_attn.q_linear."},
{"attn2.to_k.", "cross_attn.kv_linear."},
{"attn2.to_out.0.", "cross_attn.proj."},
{"ff.net.0.proj.", "mlp.fc1."},
{"ff.net.2.", "mlp.fc2."},
};
for (const auto& entry : block_map) {
if (starts_with(name, entry.first)) {
return prefix + entry.second + name.substr(entry.first.size());
}
}
static const std::vector<std::pair<std::string, std::string>> part_map = {
{"attn1.to_k.weight", "attn.qkv.weight.1"},
{"attn1.to_k.bias", "attn.qkv.bias.1"},
{"attn1.to_v.weight", "attn.qkv.weight.2"},
{"attn1.to_v.bias", "attn.qkv.bias.2"},
{"attn2.to_v.weight", "cross_attn.kv_linear.weight.1"},
{"attn2.to_v.bias", "cross_attn.kv_linear.bias.1"},
};
for (const auto& entry : part_map) {
if (name == entry.first || starts_with(name, entry.first + ".")) {
return prefix + entry.second + name.substr(entry.first.size());
}
}
return prefix + name;
}
static std::string convert_ming_image_dit_name(std::string name) {
static const std::vector<std::pair<std::string, std::string>> name_map = {
{"all_x_embedder.2-1.", "x_embedder."},
{"all_final_layer.2-1.", "final_layer."},
{"attention.norm_q.", "attention.q_norm."},
{"attention.norm_k.", "attention.k_norm."},
};
replace_with_name_map(name, name_map);
return name;
}
std::string convert_diffusion_model_name(std::string name, std::string prefix, SDVersion version) {
if (sd_version_is_sd1(version) || sd_version_is_sd2(version)) {
name = convert_diffusers_unet_to_original_sd1(name);
@@ -943,6 +1034,8 @@ std::string convert_diffusion_model_name(std::string name, std::string prefix, S
name = convert_diffusers_dit_to_original_flux(name);
} else if (sd_version_is_hunyuan_video(version)) {
name = convert_hunyuan_video_to_original_flux(name);
} else if (version == VERSION_MING_IMAGE) {
name = convert_ming_image_dit_name(name);
} else if (sd_version_is_z_image(version)) {
name = convert_diffusers_dit_to_original_lumina2(name);
} else if (sd_version_is_llada_image(version)) {
@@ -951,6 +1044,8 @@ std::string convert_diffusion_model_name(std::string name, std::string prefix, S
name = convert_other_dit_to_original_anima(name);
} else if (sd_version_is_krea2(version)) {
name = convert_diffusers_dit_to_original_krea2(name);
} else if (sd_version_is_pixart(version)) {
name = convert_diffusers_dit_to_original_pixart(name);
}
return name;
}
@@ -1604,7 +1699,7 @@ std::string convert_tensor_name(std::string name, SDVersion version) {
{
for (const auto& prefix : cond_stage_model_prefix_vec) {
if (starts_with(name, prefix)) {
name = convert_cond_stage_model_name(name.substr(prefix.size()), prefix);
name = convert_cond_stage_model_name(name.substr(prefix.size()), prefix, version);
name = prefix + name;
break;
}
+17 -3
View File
@@ -106,6 +106,7 @@ const char* model_version_to_str[] = {
"LLaDA-Image",
"ESRGAN",
"PixArt",
"Ming-Image",
};
static_assert(VERSION_COUNT == sizeof(model_version_to_str) / sizeof(model_version_to_str[0]),
@@ -1210,6 +1211,11 @@ bool StableDiffusionGGML::validate_and_load_runners() {
LOG_VERBOSE("validating model metadata");
std::set<std::string> ignore_tensors;
if (version == VERSION_MING_IMAGE) {
ignore_tensors.insert("text_encoders.llm.vision.");
ignore_tensors.insert("text_encoders.llm.linear_proj.");
ignore_tensors.insert("text_encoders.llm.backbone.lm_head.");
}
if (use_tae && !tae_preview_only) {
ignore_tensors.insert("first_stage_model.");
}
@@ -1370,6 +1376,7 @@ bool StableDiffusionGGML::build_denoiser() {
sd_version_is_anima(version) ||
sd_version_is_ernie_image(version) ||
sd_version_is_z_image(version) ||
version == VERSION_MING_IMAGE ||
sd_version_is_llada_image(version) ||
sd_version_is_boogu_image(version) ||
sd_version_is_pid(version) ||
@@ -1391,6 +1398,8 @@ bool StableDiffusionGGML::build_denoiser() {
default_flow_shift = 3.16f;
} else if (sd_version_is_mage_flow(version)) {
default_flow_shift = 6.f;
} else if (version == VERSION_MING_IMAGE) {
default_flow_shift = INFINITY;
} else if (sd_version_is_llada_image(version)) {
default_flow_shift = 1.0f; // unused: LLADA_IMAGE_SCHEDULER builds a fixed grid
} else {
@@ -1451,6 +1460,8 @@ bool StableDiffusionGGML::build_denoiser() {
} else if (sd_version_is_minimax_h3(version)) {
LOG_INFO("running in MiniMax H3 AV FLOW mode");
denoiser = std::make_shared<H3AVFlowDenoiser>(default_flow_shift, 3.f, get_latent_channel());
} else if (version == VERSION_MING_IMAGE) {
denoiser = std::make_shared<MingImageFlowDenoiser>();
} else {
LOG_INFO("running in FLOW mode");
denoiser = std::make_shared<DiscreteFlowDenoiser>();
@@ -2161,7 +2172,7 @@ std::vector<float> StableDiffusionGGML::prepare_sample_timesteps(float sigma,
if (version == VERSION_HIDREAM_O1) {
return std::vector<float>{1.0f - (t / static_cast<float>(TIMESTEPS))};
}
if (sd_version_is_z_image(version) || sd_version_is_ideogram4(version)) {
if (sd_version_is_z_image(version) || sd_version_is_ideogram4(version) || version == VERSION_MING_IMAGE) {
return std::vector<float>{1000.f - t};
}
return std::vector<float>{t};
@@ -2514,6 +2525,9 @@ sd::Tensor<float> StableDiffusionGGML::sample(const std::shared_ptr<DiffusionMod
condition.c_token_types.empty() ? nullptr : &condition.c_token_types,
condition.c_vinput_mask.empty() ? nullptr : &condition.c_vinput_mask,
condition.c_image_embeds.empty() ? nullptr : &condition.c_image_embeds};
} else if (version == VERSION_MING_IMAGE) {
diffusion_params.extra = MingImageDiffusionExtra{
condition.extra_c_crossattns.empty() ? nullptr : &condition.extra_c_crossattns[0]};
} else if (sd_version_is_llada_image(version)) {
diffusion_params.extra = LLaDAImageDiffusionExtra{
condition.extra_c_crossattns.empty() ? nullptr : &condition.extra_c_crossattns[0]};
@@ -2748,7 +2762,7 @@ int StableDiffusionGGML::get_diffusion_model_down_factor() {
if (sd_version_is_dit(version)) {
if (sd_version_is_sensenova_u1(version)) {
down_factor = 32;
} else if (version == VERSION_QWEN_IMAGE_2_1 || sd_version_is_wan(version) || sd_version_is_lingbot_video(version) || sd_version_is_minimax_h3(version) || sd_version_is_pixart(version)) {
} else if (version == VERSION_QWEN_IMAGE_2_1 || version == VERSION_MING_IMAGE || sd_version_is_wan(version) || sd_version_is_lingbot_video(version) || sd_version_is_minimax_h3(version) || sd_version_is_pixart(version)) {
down_factor = 2;
} else {
down_factor = 1;
@@ -2796,7 +2810,7 @@ int StableDiffusionGGML::get_latent_channel() {
}
int StableDiffusionGGML::get_image_channels() const {
return version == VERSION_QWEN_IMAGE_LAYERED || version == VERSION_QWEN_IMAGE_2_1 ? 4 : 3;
return version == VERSION_QWEN_IMAGE_LAYERED || version == VERSION_QWEN_IMAGE_2_1 || version == VERSION_MING_IMAGE ? 4 : 3;
}
int StableDiffusionGGML::get_image_seq_len(int h, int w) {
+5
View File
@@ -465,6 +465,11 @@ namespace sd::pipeline {
// states with a zeroed prompt mask, so no extra text encode is needed.
uncond.c_crossattn = cond.c_crossattn;
uncond.c_vector = sd::Tensor<float>::zeros_like(cond.c_vector);
} else if (sd->version == VERSION_MING_IMAGE) {
uncond.c_crossattn = sd::Tensor<float>::zeros_like(cond.c_crossattn);
for (const auto& extra : cond.extra_c_crossattns) {
uncond.extra_c_crossattns.push_back(sd::Tensor<float>::zeros_like(extra));
}
} else if (sd_version_is_sensenova_u1(sd->version)) {
auto* sensenova_conditioner = static_cast<SenseNovaU1Conditioner*>(sd->cond_stage_model.get());
uncond = sensenova_conditioner->get_unconditional_condition(request->negative_prompt);
+7 -1
View File
@@ -24,6 +24,7 @@
#include "model/diffusion/llada_image.hpp"
#include "model/diffusion/ltxv.hpp"
#include "model/diffusion/mage_flow.hpp"
#include "model/diffusion/ming_image.hpp"
#include "model/diffusion/minimax_h3.hpp"
#include "model/diffusion/minit2i.hpp"
#include "model/diffusion/mmdit.hpp"
@@ -374,6 +375,11 @@ namespace sd::model_builders {
tensor_storage_map,
"model.diffusion_model",
weight_manager);
} else if (version == VERSION_MING_IMAGE) {
result.conditioner = std::make_shared<MingImageEmbedder>(ctx.backends.runtime_backend(SDBackendModule::TE),
tensor_storage_map, weight_manager, tokenizers);
result.diffusion = std::make_shared<MingImage::MingImageRunner>(ctx.backends.runtime_backend(SDBackendModule::DIFFUSION),
tensor_storage_map, "model.diffusion_model", weight_manager);
} else if (sd_version_is_z_image(version)) {
result.conditioner = std::make_shared<LLMEmbedder>(ctx.backends.runtime_backend(SDBackendModule::TE),
tensor_storage_map,
@@ -579,7 +585,7 @@ namespace sd::model_builders {
weight_manager);
if (sd_version_is_pixart(version)) {
// Alpha-512 and Sigma share tensor layouts; Alpha-512 needs an explicit scale override.
if (tensor_storage_map.count("model.diffusion_model.adaln_single.emb.resolution_embedder.linear_1.weight") != 0) {
if (tensor_storage_map.count("model.diffusion_model.csize_embedder.mlp.0.weight") != 0) {
model->scale_factor = 0.18215f;
}
for (const auto& [key, value] : parse_key_value_args(sd_ctx_params->model_args, "model arg")) {
+40 -1
View File
@@ -1369,6 +1369,42 @@ struct DiscreteFlowDenoiser : public Denoiser {
}
};
struct MingImageFlowDenoiser : DiscreteFlowDenoiser {
float resolution_shift = std::exp(1.35f);
MingImageFlowDenoiser()
: DiscreteFlowDenoiser(INFINITY) {}
float sigma_min() override { return 0.f; }
float sigma_max() override { return 1.f; }
float t_to_sigma(float t) override {
return time_snr_shift(std::isfinite(shift) ? shift : resolution_shift, (t + 1.f) / 1000.f);
}
std::vector<float> get_sigmas(uint32_t n, int image_seq_len, scheduler_t scheduler, SDVersion version, const char* extra_sample_args = nullptr) override {
const float tokens = static_cast<float>(image_seq_len) / 4.f;
const float mu = tokens >= 4096.f ? 1.35f : 0.5f + (tokens - 256.f) * (1.15f - 0.5f) / (4096.f - 256.f);
resolution_shift = std::exp(mu);
if (scheduler != DISCRETE_SCHEDULER) {
return DiscreteFlowDenoiser::get_sigmas(n, image_seq_len, scheduler, version, extra_sample_args);
}
if (n == 0) {
return {};
}
const float factor = std::isfinite(shift) ? shift : resolution_shift;
std::vector<float> sigmas;
sigmas.reserve(n + 1);
for (uint32_t i = 0; i < n; ++i) {
const float t = n == 1 ? 1.f : 1.f - static_cast<float>(i) / static_cast<float>(n - 1);
sigmas.push_back(time_snr_shift(factor, t));
}
// Upstream sets sigma_min to zero before linspace, then appends the terminal zero.
sigmas.push_back(0.f);
return sigmas;
}
};
struct H3AVFlowDenoiser : public DiscreteFlowDenoiser {
int64_t video_channels;
float audio_shift;
@@ -1783,7 +1819,10 @@ static sd::Tensor<float> sample_euler(denoise_cb_t model,
const std::vector<float>& sigmas) {
int steps = static_cast<int>(sigmas.size()) - 1;
for (int i = 0; i < steps; i++) {
float sigma = sigmas[i];
float sigma = sigmas[i];
if (sigma == sigmas[i + 1]) {
continue;
}
auto denoised_opt = model(x, sigma, i + 1);
if (denoised_opt.pred.empty()) {
return {};
+1 -1
View File
@@ -301,7 +301,7 @@ bool T5UniGramTokenizer::encode(const std::string& input, std::vector<int>& resu
result = std::move(tokens);
return true;
}
auto splited_texts = split_with_special_tokens(normalized, special_tokens);
auto splited_texts = split_with_special_tokens(normalized, special_tokens);
if (splited_texts.empty()) {
splited_texts.push_back(normalized); // for empty string
}