Compare commits

...
Author SHA1 Message Date
Wagner Bruna 008ca5b492 feat: restore legacy fp8 handling when building with upstream ggml (#2001) 2026-09-21 00:48:30 +08:00
leejet 137f7409bb feat: add Qwen Image 2.1 support (#1994) 2026-09-20 22:51:21 +08:00
leejet 1330cebae8 feat: support building with upstream ggml (#1999) 2026-09-19 22:25:01 +08:00
leejet 17860c0e45 perf: parallelize host tensor elementwise and broadcast ops (#1998) 2026-09-19 21:46:16 +08:00
leejet 275ab58e01 perf: reduce CPU overhead in graph execution and sampling (#1997) 2026-09-19 18:32:58 +08:00
Fabrice Aneche d32b4e893b fix: prevent clip_preprocess center crop from exceeding the resized image (#1995) 2026-09-19 18:08:34 +08:00
leejet 9982c9caae fix: propagate CUDA driver dependency to shared library consumers 2026-09-19 18:04:59 +08:00
leejet 3e037a81e4 perf: accelerate VAE direct 3D convolutions (#1996) 2026-09-19 17:50:01 +08:00
leejet 2ea8aff7ef perf: update ggml for faster direct convolutions (#1993) 2026-09-19 00:32:24 +08:00
leejet adcac69650 perf: pad small attention heads to 64 for MMA Flash Attention (#1992) 2026-09-18 23:59:58 +08:00
leejet 656a1354c3 refactor: remove obsolete unused tensor filtering (#1984) 2026-09-18 23:41:37 +08:00
Lin Xuhao 269e726015 fix: honor flash attention flag in LLM text encoder attention (#1987) 2026-09-18 23:41:24 +08:00
leejet cc515a01f9 perf: eliminate temporary allocations in Philox rounds (#1982) 2026-09-17 02:09:12 +08:00
leejet 3161505fe8 fix: remove vision_model. from ununsed tensors (#1983) 2026-09-17 02:08:24 +08:00
leejet 59c23bce0d fix: use tokenizer-specific pre-tokenization rules (#1975) 2026-09-15 02:37:52 +08:00
Wagner Bruna 07a85c74cb feat: support Brownian tree noise in all noise injection samplers (#1899) 2026-09-15 02:37:31 +08:00
leejet f9ddc0f388 refactor: require external Gemma 2 and GPT-OSS tokenizers (#1974) 2026-09-15 01:25:13 +08:00
leejet 4964abdfc5 feat: support external Hugging Face tokenizer JSON files (#1973) 2026-09-15 00:27:39 +08:00
Санька Четвёртыйandleejet 42d6c0ab92 feat: Add generation parameters into video metadata (#1901)
Co-authored-by: leejet <leejet714@gmail.com>
2026-09-14 00:01:09 +08:00
leejet 5a5400bf0c fix: resolve MSVC narrowing conversion warnings (#1969) 2026-09-13 23:45:54 +08:00
fszontagh ca37fad89a fix: validate vision projector output dim against LLM hidden size (#1918) 2026-09-13 23:42:46 +08:00
Georgeandleejet 0bd72f075a feat: add Wan2.2 S2V (audio+img-to-video) support (#1925)
Co-authored-by: leejet <leejet714@gmail.com>
2026-09-13 23:33:50 +08:00
fszontagh 4a7da26b73 fix: bound plain-text runs in parse_prompt_attention regex (#1919) 2026-09-13 23:29:19 +08:00
leejet 9a977388a8 fix: guard GPU memory capacity and propagate encoding failures (#1958) 2026-09-13 21:35:46 +08:00
leejet 44dd13716d feat: preserve explicit backend assignments during auto-fit (#1967) 2026-09-13 17:40:38 +08:00
leejet 7f410a3793 feat: add linear and attention scale overrides (#1964) 2026-09-12 01:41:28 +08:00
leejet 5ebce93342 fix: reuse graph plans when scale parameters change (#1963) 2026-09-12 01:04:25 +08:00
Maphist0 7f986a9d73 feat: add SenseNova U1.5 support (#1935) 2026-09-12 00:39:59 +08:00
fszontagh e06b205384 feat: expose the loaded model version name through the public API (#1962) 2026-09-11 23:39:56 +08:00
vmobilis 3191b23d4b fix: handle invalid option numbers (#1961) 2026-09-11 23:16:39 +08:00
LED-M e95ab96997 fix: preserve BF16 embedding weights for get_rows (#1959) 2026-09-11 23:09:45 +08:00
Hmission b68d58624d fix: enable VAE decode tiling fallback without auto-fit (#1932) 2026-09-11 01:34:56 +08:00
leejet 14eddb32b1 refactor: split generation pipeline out of stable-diffusion.cpp (#1957) 2026-09-11 00:57:14 +08:00
stduhpf 469fc49bb7 docs: reflect GGML_MAX_NAME value change in rpc docs (and in ggml_extend assert) (#1950) 2026-09-10 23:55:02 +08:00
180 changed files with 109591 additions and 8679 deletions
+3 -1
View File
@@ -6,7 +6,9 @@ body:
- type: markdown
attributes:
value: |
Please use this template and include as many details as possible to help us reproduce and fix the issue.
Before submitting a bug report, please read the [Troubleshooting guide](https://github.com/leejet/stable-diffusion.cpp/blob/master/docs/troubleshooting.md) and try the steps relevant to your problem.
If the problem persists, complete this form and include what you tried and the results, along with enough details to help us reproduce and fix the issue.
- type: textarea
id: commit
attributes:
+4
View File
@@ -0,0 +1,4 @@
contact_links:
- name: Troubleshooting
url: https://github.com/leejet/stable-diffusion.cpp/blob/master/docs/troubleshooting.md
about: Read the troubleshooting guide first. If the problem persists, submit a bug report.
+2 -1
View File
@@ -118,7 +118,8 @@ documentation.
6. Run the narrowest useful build, test, or inspection command available.
Follow `CONTRIBUTING.md` for formatting, naming, PR expectations, dependency
update policy, and security rules.
update policy, and security rules. For tokenizer additions, follow its embedded-data
allowlist and default to an external `tokenizer.json`.
---
+16 -12
View File
@@ -95,6 +95,8 @@ option(SD_MUSA "sd: musa backend" OFF)
option(SD_BUILD_SHARED_LIBS "sd: build shared libs" OFF)
option(SD_BUILD_SHARED_GGML_LIB "sd: build ggml as a separate shared lib" OFF)
option(SD_USE_SYSTEM_GGML "sd: use system-installed GGML library" OFF)
option(SD_USE_UPSTREAM_GGML "sd: build with upstream GGML instead of the patched GGML extensions" OFF)
set(SD_GGML_SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/ggml" CACHE PATH "sd: ggml source directory (also supplies private headers for system ggml)")
#option(SD_BUILD_SERVER "sd: build server example" ON)
set(CMAKE_C_STANDARD 11)
@@ -229,6 +231,8 @@ file(GLOB SD_LIB_SOURCES CONFIGURE_DEPENDS
"src/model/*/*.h"
"src/model/*/*.cpp"
"src/model/*/*.hpp"
"src/pipeline/*.h"
"src/pipeline/*.cpp"
"src/runtime/*.h"
"src/runtime/*.cpp"
"src/runtime/*.hpp"
@@ -292,6 +296,8 @@ endif()
if(MSVC)
target_compile_options(${SD_LIB} PRIVATE $<$<COMPILE_LANGUAGE:CXX>:/bigobj>)
# ggml backends can throw C++ exceptions through their C API.
target_compile_options(${SD_LIB} PRIVATE $<$<AND:$<COMPILE_LANGUAGE:CXX>,$<CXX_COMPILER_ID:MSVC>>:/EHsc->)
endif()
if(APPLE)
@@ -321,23 +327,21 @@ if (NOT SD_USE_SYSTEM_GGML)
endif()
# deps
# Only add ggml if it hasn't been added yet
if (NOT TARGET ggml)
if (SD_USE_SYSTEM_GGML)
find_package(ggml REQUIRED)
if (NOT ggml_FOUND)
message(FATAL_ERROR "System-installed GGML library not found.")
endif()
add_library(ggml ALIAS ggml::ggml)
else()
add_subdirectory(ggml)
endif()
endif()
include(cmake/ggml.cmake)
add_subdirectory(thirdparty)
target_sources(${SD_LIB} PRIVATE $<TARGET_OBJECTS:zip>)
target_link_libraries(${SD_LIB} PUBLIC ggml)
find_package(Threads REQUIRED)
target_link_libraries(${SD_LIB} PRIVATE Threads::Threads)
target_link_libraries(${SD_LIB} PRIVATE onig sd-utf8proc)
if (SD_CUDA)
find_package(CUDAToolkit REQUIRED)
# Keep the driver stub on downstream link lines when no driver is installed.
target_link_libraries(${SD_LIB} PUBLIC CUDA::cuda_driver)
set_property(SOURCE src/core/ggml_extend_backend.cpp APPEND PROPERTY COMPILE_DEFINITIONS SD_USE_CUDA)
endif()
target_include_directories(${SD_LIB} PUBLIC . src include)
target_include_directories(${SD_LIB} PRIVATE src/core)
target_include_directories(${SD_LIB} PUBLIC . thirdparty)
+15
View File
@@ -46,6 +46,21 @@ Some older code in the project may not fully follow the current conventions. Ple
When adding or modifying model implementations, follow the model config and weight detection conventions in [docs/model_config.md](docs/model_config.md).
## Tokenizer Data
New model integrations must use an external `tokenizer.json` by default. Do not
embed new vocabularies or merge tables solely for less widely used models;
these tables increase the binary size for every user.
The embedded-data allowlist is CLIP, T5/UMT5, Qwen 2/3, Mistral, and Gemma 3/4.
Models may reuse an existing embedded tokenizer when its vocabulary and behavior
match their text encoder. Gemma 2 and GPT-OSS require external JSON files.
Adding to this allowlist requires maintainer approval, supported by the model's
usage, reuse across models, and measured binary-size cost. Document the matching
JSON and CLI option for models that require an external tokenizer, and fail
initialization clearly when it is missing.
## AI-Assisted Contributions
AI tools may be used to assist development, but contributors are responsible for the quality and correctness of the submitted code.
+4
View File
@@ -15,6 +15,7 @@ API and command-line option may change frequently.***
## 🔥Important News
* **2026/09/20** 🚀 stable-diffusion.cpp adds **Day-0 support for Qwen-Image-2.1**
* **2026/08/20** 🚀 stable-diffusion.cpp now supports **LTX-2.5**
* **2026/08/04** 🚀 stable-diffusion.cpp adds **Day-1 support for MiniMax-H3**
* **2026/06/25** 🚀 stable-diffusion.cpp now supports **Krea2**
@@ -47,10 +48,12 @@ API and command-line option may change frequently.***
- [Chroma](./docs/chroma.md)
- [Chroma1-Radiance](./docs/chroma_radiance.md)
- [Qwen Image](./docs/qwen_image.md)
- [Qwen Image 2.1](./docs/qwen_image_2.1.md)
- [PiD](./docs/pid.md)
- [LongCat Image](./docs/longcat_image.md)
- [Z-Image](./docs/z_image.md)
- [MiniT2I](./docs/minit2i.md)
- [SenseNova U1.5](./docs/sensenova_u1.md)
- [Ovis-Image](./docs/ovis_image.md)
- [Anima](./docs/anima.md)
- [ERNIE-Image](./docs/ernie_image.md)
@@ -147,6 +150,7 @@ For runtime and parameter backend placement, see the [backend selection guide](.
## More Guides
- [Troubleshooting](./docs/troubleshooting.md)
- [Backend selection](./docs/backend.md)
- [RPC](./docs/rpc.md)
- [LoRA](./docs/lora.md)
Binary file not shown.

After

Width:  |  Height:  |  Size: 634 KiB

+27
View File
@@ -0,0 +1,27 @@
if(NOT TARGET ggml AND NOT TARGET ggml::ggml)
if(SD_USE_SYSTEM_GGML)
find_package(ggml REQUIRED)
else()
add_subdirectory("${SD_GGML_SOURCE_DIR}" "${CMAKE_CURRENT_BINARY_DIR}/ggml")
endif()
endif()
if(NOT TARGET ggml)
add_library(ggml ALIAS ggml::ggml)
endif()
get_target_property(sd_ggml_imported ggml IMPORTED)
if(sd_ggml_imported)
set(sd_ggml_private_include "${SD_GGML_SOURCE_DIR}/src")
else()
get_target_property(sd_ggml_private_include ggml SOURCE_DIR)
endif()
if(NOT EXISTS "${sd_ggml_private_include}/ggml-impl.h")
message(FATAL_ERROR "Set SD_GGML_SOURCE_DIR to the source tree matching the selected ggml library (ggml-impl.h is required).")
endif()
target_include_directories(${SD_LIB} PRIVATE "${sd_ggml_private_include}")
set_property(TARGET ${SD_LIB} PROPERTY SD_GGML_PRIVATE_INCLUDE_DIR "${sd_ggml_private_include}")
if(SD_USE_UPSTREAM_GGML)
target_compile_definitions(${SD_LIB} PUBLIC SD_USE_UPSTREAM_GGML)
message(WARNING "Using upstream GGML: INT8 tensorwise/convrot is disabled and FP8 weights are converted to F16 at load time. Some operators may be unsupported and performance may be lower than with patched GGML.")
endif()
+9 -1
View File
@@ -10,6 +10,10 @@ set(SD_BIN_DIR "@PACKAGE_SD_BIN_INSTALL_DIR@")
include(CMakeFindDependencyMacro)
find_dependency(ggml REQUIRED HINTS "${SD_LIB_DIR}/cmake")
find_dependency(Threads REQUIRED)
if(@SD_CUDA@)
find_dependency(CUDAToolkit REQUIRED)
endif()
if(NOT TARGET stable-diffusion)
find_library(stable-diffusion_LIBRARY stable-diffusion
@@ -22,12 +26,16 @@ if(NOT TARGET stable-diffusion)
set_target_properties(stable-diffusion
PROPERTIES
INTERFACE_INCLUDE_DIRECTORIES "${SD_INCLUDE_DIR}"
INTERFACE_LINK_LIBRARIES "ggml::ggml"
INTERFACE_LINK_LIBRARIES "ggml::ggml;Threads::Threads"
IMPORTED_LINK_INTERFACE_LANGUAGES "CXX"
IMPORTED_LOCATION "${stable-diffusion_LIBRARY}"
INTERFACE_COMPILE_FEATURES "c_std_11;cxx_std_17"
POSITION_INDEPENDENT_CODE ON)
if(@SD_CUDA@)
set_property(TARGET stable-diffusion APPEND PROPERTY INTERFACE_LINK_LIBRARIES CUDA::cuda_driver)
endif()
if(SD_SHARED_LIB)
target_compile_definitions(stable-diffusion
INTERFACE SD_BUILD_SHARED_LIB)
+1 -1
View File
@@ -7,5 +7,5 @@ Name: stable-diffusion
Description: Diffusion model(SD,Flux,Wan,Qwen Image,Z-Image,...) inference in pure C/C++
Version: @SDCPP_BUILD_VERSION@
Libs: -L${libdir} -lstable-diffusion
Libs.private: -lggml -lggml-base
Libs.private: -lggml -lggml-base @CMAKE_THREAD_LIBS_INIT@
Cflags: -I${includedir}
+33 -14
View File
@@ -5,7 +5,8 @@
- `--backend` selects the runtime backend used to execute model graphs.
- `--params-backend` selects where model parameters are kept.
If `--params-backend` is not set, parameters use the same backend as their module runtime backend.
If `--params-backend` is not set, auto-fit chooses parameter placement. With
`--auto-fit off`, parameters use the same backend as their module runtime backend.
## Syntax
@@ -129,17 +130,21 @@ warning.
## Automatic placement (`--auto-fit on|off`)
`--auto-fit` requires `on` or `off` and defaults to `on` when omitted.
Explicit `--backend` or `--params-backend` assignments disable auto-fit,
Explicit `--params-backend` assignments disable auto-fit,
regardless of argument order, even with `--auto-fit on`.
When enabled, auto-fit uses one GPU for `diffusion` / `te` / `vae` computation. It chooses
the GPU with the largest available memory budget (the first device on a tie),
then derives parameter placements from the model metadata and the remaining
memory budgets. The chosen backend specifications are printed.
Auto-fit preserves explicit `--backend` assignments, including per-module
assignments and device lists. For modules without a runtime assignment, it chooses
the GPU with the largest available memory budget (the first device on a tie).
It then derives parameter placements from the model metadata, each module's
compute devices, and the remaining memory budgets. The chosen backend
specifications are printed.
```shell
sd-cli -m model.safetensors -p "a cat" --auto-fit on
sd-cli -m model.safetensors -p "a cat" --auto-fit on --max-vram cuda0=8,cuda1=14
sd-cli -m model.safetensors -p "a cat" --backend cuda0
sd-cli -m model.safetensors -p "a cat" --backend diffusion=cuda0,te=cpu,vae=cuda1
sd-cli -m model.safetensors -p "a cat" --auto-fit off
```
@@ -149,11 +154,16 @@ GiB", and with no budget set each device's free memory minus a 512 MiB margin
is used. These resolved GPU budgets, including the safety margin, also drive
the runner's graph-cut capacity checks.
Runtime capacity checks also leave 512 MiB of currently free device memory for
backend scratch buffers and pipelines, including with explicit backend assignments.
They cap stale free-memory reports by the device's total memory minus tracked
resident allocations and reject reports that exceed the device's total memory.
Components are considered in `diffusion`, `te`, `vae` order so that repeatedly
used diffusion weights have priority. Each component's weights use the first
storage location with enough remaining budget:
1. The main GPU, leaving estimated space for computation and weight staging.
1. The component's compute GPU, leaving estimated space for computation and weight staging.
2. CPU RAM, reserving the larger of 2 GiB or 10% of available RAM for other work.
3. Another GPU, choosing the one with the largest remaining budget that fits.
4. Disk, reloading weights on demand.
@@ -170,10 +180,17 @@ weight to be copied again at every step.
RAM and GPU budgets are shared across components. Each component uses a single
parameter backend; several other GPUs' capacities are not combined to store
one component. If available RAM cannot be queried, RAM residency is skipped.
Other GPUs store weights only: weights are copied to the main GPU for execution.
Auto-fit does not select multi-GPU layer/row computation, so `--split-mode` does
not change its placements. Use explicit backend assignments for multi-GPU
computation.
Weights stored on another GPU are copied to the component's compute devices for
execution. CPU modules use RAM or disk. Compute reserves and cache priority are
accounted for separately on each device, so a CPU module does not reserve GPU
space. Storage on another module's GPU also leaves room for that module's work.
Auto-fit does not select multi-GPU layer/row computation itself. Explicit device
lists and `--split-mode` still control that computation. Before the runners have
built their split plans, auto-fit conservatively counts the full component size
on each listed GPU when checking residency and cache space. This can offload
parameters even when a split layout would fit; use `--auto-fit off` to keep the
default split-device parameter placement.
For example, a diffusion model whose full weights exceed the main GPU's budget
can use `--backend diffusion=cuda0 --params-backend diffusion=cpu` when RAM is
@@ -188,8 +205,9 @@ weights, compute buffers and caches must
still fit the runner's capacity checks. Offloading weights does not guarantee
that every resolution or frame count will fit, and auto-fit does not change a
component to CPU computation solely because its full weights exceed VRAM.
If a VAE decode fails, auto-fit retries with spatial tiling; supported video
decoders try temporal tiling first and can then add spatial tiling.
If a VAE decode fails, decoding retries with spatial tiling even when `--auto-fit`
is off; supported video decoders try temporal tiling first and can then add
spatial tiling. Spatial retries use half-size tiles along each latent dimension.
## Modules
@@ -291,6 +309,7 @@ The example CLI/server still accepts these older CPU placement flags as compatib
Because this default is inserted first, later explicit `--params-backend` entries can still override it, for example `--offload-to-cpu --params-backend te=disk` keeps non-TE parameters on CPU and reloads TE parameters from disk.
Library callers should set `backend` and `params_backend` directly. `sd_ctx_params_init()`
enables `auto_fit` by default; nonempty `backend` or `params_backend` assignments disable it.
enables `auto_fit` by default; a nonempty `params_backend` assignment disables it.
The `backend` assignment constrains auto-fit's compute placement.
The old CPU/offload fields are no longer part of the C API. Explicit `--backend` and
`--params-backend` assignments are preferred for new commands.
+34
View File
@@ -16,6 +16,40 @@ git submodule init
git submodule update
```
## Selecting a GGML source tree
By default, sd.cpp builds the patched GGML submodule in `ggml/`. To build with
an upstream GGML checkout instead, enable `SD_USE_UPSTREAM_GGML` and set
`SD_GGML_SOURCE_DIR`:
```shell
cmake -S . -B build-upstream -DSD_USE_UPSTREAM_GGML=ON -DSD_GGML_SOURCE_DIR=../ggml-upstream
cmake --build build-upstream --config Release
```
The selected source tree supplies both the library and its private headers.
Backend options such as `-DSD_CUDA=ON` apply to the selected tree as usual.
`SD_USE_UPSTREAM_GGML` defaults to `OFF`, which enables the patched GGML
extensions. Set it to `ON` when using upstream GGML; it selects the compatibility
mode and does not download or replace the GGML source tree. Upstream mode keeps
the original FP8 safetensors handling: FP8 tensors are converted to F16 at load
time (one byte per element in the file, two in RAM and VRAM). INT8
tensorwise/convrot is disabled and its model files are rejected with an explicit
error. FP8 GGUF files, FP8 weight type requests and tensor type rules are also
rejected; no automatic conversion is performed.
Upstream GGML may lack some operators and performance optimizations provided by
the patched version. A warning is emitted during CMake configuration and when
creating an inference context. Ordinary floating-point and shared GGML
quantization types remain available, subject to backend operator support.
`SD_USE_SYSTEM_GGML=ON` instead links an installed GGML CMake package, located
with `ggml_DIR` or `CMAKE_PREFIX_PATH`. In that mode, `SD_GGML_SOURCE_DIR` must
point to the matching source tree for private headers. The installed library
must use the same ABI settings as sd.cpp, including `GGML_MAX_NAME`.
Set `SD_USE_UPSTREAM_GGML=ON` as well if the installed package is upstream GGML.
## WebP and WebM Support in Examples
The example applications (`examples/cli` and `examples/server`) use `libwebp` to support WebP image I/O, and `examples/cli` can also use `libwebm` for `.webm` video output. Both are enabled by default. WebM output currently reuses `libwebp` to encode each frame as VP8 before muxing with `libwebm`.
+3
View File
@@ -18,6 +18,9 @@ at one byte per element in RAM and VRAM. Backends that cannot multiply FP8
weights directly cast only the active layer to a temporary BF16 tensor during
execution; the loader does not expand the entire checkpoint to BF16.
With `SD_USE_UPSTREAM_GGML=ON`, FP8 tensors are converted to F16 at load time
instead (two bytes per element in RAM and VRAM).
Use `ideogram4_fp8.safetensors` and `ideogram4_uncond_fp8.safetensors` directly
with `--diffusion-model` and `--uncond-diffusion-model`, respectively.
+3
View File
@@ -2,6 +2,9 @@
sd.cpp can load and execute ComfyUI `int8_tensorwise` safetensors with `convrot` metadata directly. The stored INT8 weights are not converted to another weight type at load time.
This requires the INT8 tensorwise/convrot extensions in the patched GGML.
Builds with `SD_USE_UPSTREAM_GGML=ON` reject these files during loading.
## Checkpoint format
Each quantized linear module contains the following tensors:
+6 -2
View File
@@ -12,13 +12,17 @@ Lens uses a Lens diffusion transformer, the FLUX.2 VAE, and GPT-OSS-20B as the L
- safetensors: https://huggingface.co/black-forest-labs/FLUX.2-dev/tree/main
- Download GPT-OSS-20B
- gguf: https://huggingface.co/unsloth/gpt-oss-20b-GGUF/tree/main
- Download GPT-OSS-20B tokenizer.json
- https://huggingface.co/openai/gpt-oss-20b/tree/main
Lens and Lens Turbo require an external GPT-OSS `tokenizer.json` matching the text encoder checkpoint. Save it as `tokenizer_gpt_oss.json` and pass it with `--tokenizer`; the tokenizer is not embedded in sd.cpp. See [JSON tokenizers](tokenizers.md) for CLI and C API usage.
## Examples
### Lens
```
.\bin\Release\sd-cli.exe --diffusion-model ..\models\diffusion_models\lens_bf16.safetensors --llm "..\models\text_encoders\gpt-oss-20b-UD-Q8_K_XL.gguf" --vae ..\models\vae\flux2_ae.safetensors --cfg-scale 5.0 -p "A crystal dragon soaring through an aurora borealis sky, its entire body made of transparent faceted crystal refracting the green and purple aurora light into rainbow spectra, ice particles trailing from its wings, high fantasy digital art" --diffusion-fa -v
.\bin\Release\sd-cli.exe --diffusion-model ..\models\diffusion_models\lens_bf16.safetensors --llm "..\models\text_encoders\gpt-oss-20b-UD-Q8_K_XL.gguf" --tokenizer ..\models\tokenizers\tokenizer_gpt_oss.json --vae ..\models\vae\flux2_ae.safetensors --cfg-scale 5.0 -p "A crystal dragon soaring through an aurora borealis sky, its entire body made of transparent faceted crystal refracting the green and purple aurora light into rainbow spectra, ice particles trailing from its wings, high fantasy digital art" --diffusion-fa -v
```
<img width="256" alt="Lens example" src="../assets/lens/example.png" />
@@ -26,7 +30,7 @@ Lens uses a Lens diffusion transformer, the FLUX.2 VAE, and GPT-OSS-20B as the L
### Lens Turbo
```
.\bin\Release\sd-cli.exe --diffusion-model ..\models\diffusion_models\lens_turbo_bf16.safetensors --llm "..\models\text_encoders\gpt-oss-20b-UD-Q8_K_XL.gguf" --vae ..\models\vae\flux2_ae.safetensors --cfg-scale 1.0 -p "A crystal dragon soaring through an aurora borealis sky, its entire body made of transparent faceted crystal refracting the green and purple aurora light into rainbow spectra, ice particles trailing from its wings, high fantasy digital art" --diffusion-fa -v --steps 4
.\bin\Release\sd-cli.exe --diffusion-model ..\models\diffusion_models\lens_turbo_bf16.safetensors --llm "..\models\text_encoders\gpt-oss-20b-UD-Q8_K_XL.gguf" --tokenizer ..\models\tokenizers\tokenizer_gpt_oss.json --vae ..\models\vae\flux2_ae.safetensors --cfg-scale 1.0 -p "A crystal dragon soaring through an aurora borealis sky, its entire body made of transparent faceted crystal refracting the green and purple aurora light into rainbow spectra, ice particles trailing from its wings, high fantasy digital art" --diffusion-fa -v --steps 4
```
<img width="256" alt="Lens Turbo example" src="../assets/lens/turbo_example.png" />
+1 -1
View File
@@ -27,7 +27,7 @@ Using `--offload-to-cpu` allows you to offload weights to the CPU, saving VRAM w
## Use params backend to reduce VRAM or RAM usage.
`--params-backend` controls where model parameters are kept. If it is not set, parameters use the same backend as `--backend`, so a GPU runtime backend also keeps parameters in VRAM.
`--params-backend` controls where model parameters are kept. If it is not set, auto-fit chooses parameter placement while preserving `--backend`. With `--auto-fit off`, parameters use the same backend as `--backend`, so a GPU runtime backend also keeps parameters in VRAM.
Use CPU params to reduce VRAM usage:
+5 -1
View File
@@ -11,6 +11,8 @@ In stable-diffusion.cpp, PiD currently runs as an image edit pipeline: provide a
- safetensors: https://huggingface.co/Comfy-Org/PixelDiT/tree/main/diffusion_models
- Download Gemma 2 2B
- safetensors: https://huggingface.co/Comfy-Org/PixelDiT/tree/main/text_encoders
- Download Gemma 2 2B tokenizer.json
- https://huggingface.co/google/gemma-2-2b/tree/main
- Download the VAE that matches the PiD checkpoint backbone
- safetensors: https://huggingface.co/nvidia/PiD/tree/main/checkpoints
- Flux / Z-Image PiD: use the Flux VAE and pass `--vae-format flux`
@@ -20,10 +22,12 @@ In stable-diffusion.cpp, PiD currently runs as an image edit pipeline: provide a
The official PiD model card should be checked before use. At the time of the initial PiD release, the official weights are under the NSCLv1 non-commercial license.
PiD and PiD 1.5 require an external Gemma 2 `tokenizer.json` matching the text encoder checkpoint. Save it as `tokenizer_gemma2.json` and pass it with `--tokenizer`; the tokenizer is not embedded in sd.cpp. See [JSON tokenizers](tokenizers.md) for CLI and C API usage.
## Examples
```
.\bin\Release\sd-cli.exe --diffusion-model ..\models\diffusion_models\pid_flux1_512_to_2048_4step_bf16.safetensors --llm "..\models\text_encoders\gemma_2_2b_it_elm_bf16.safetensors" --vae ..\models\vae\ae.sft --vae-format flux --cfg-scale 1.0 -p "a lovely cat" -r ..\assets\ernie_image\turbo_example.png --diffusion-fa -v --steps 4 -H 2048 -W 2048 --rng cpu
.\bin\Release\sd-cli.exe --diffusion-model ..\models\diffusion_models\pid_flux1_512_to_2048_4step_bf16.safetensors --llm "..\models\text_encoders\gemma_2_2b_it_elm_bf16.safetensors" --tokenizer ..\models\tokenizers\tokenizer_gemma2.json --vae ..\models\vae\ae.sft --vae-format flux --cfg-scale 1.0 -p "a lovely cat" -r ..\assets\ernie_image\turbo_example.png --diffusion-fa -v --steps 4 -H 2048 -W 2048 --rng cpu
```
Before:
+41
View File
@@ -0,0 +1,41 @@
# How to Use
Qwen Image 2.1 supports text-to-image generation and image editing, using Qwen3-VL-8B as the text encoder and its own VAE.
## Download weights
- Download Qwen Image 2.1
- safetensors: https://huggingface.co/Comfy-Org/Qwen-Image-2.1/tree/main/diffusion_models
- gguf: https://huggingface.co/leejet/Qwen-Image-2.1-GGUF/tree/main
- Download vae
- safetensors: https://huggingface.co/Comfy-Org/Qwen-Image-2.1/tree/main/vae
- Download Qwen3-VL-8B-Instruct
- safetensors (BF16 or INT8 convrot): https://huggingface.co/Comfy-Org/Qwen-Image-2.1/tree/main/text_encoders
- gguf: https://huggingface.co/Qwen/Qwen3-VL-8B-Instruct-GGUF/tree/main
- For image editing with a GGUF text encoder, also download `mmproj-Qwen3VL-8B-Instruct-F16.gguf` from the same repository and pass it with `--llm_vision`.
Use `qwen_image_2.1_vae_bf16.safetensors` with this model. The earlier Qwen Image and Wan 2.2 VAE weights are not interchangeable with the Qwen Image 2.1 VAE weights.
## Examples
Run the following commands from the build directory. Use image dimensions divisible by 32. The resolution-dependent flow schedule is selected automatically.
### Text to image
```powershell
.\bin\Release\sd-cli.exe --diffusion-model ..\models\diffusion_models\qwen_image_2.1_int8_convrot.safetensors --vae ..\models\vae\qwen_image_2.1_vae_bf16.safetensors --llm ..\models\text_encoders\Qwen3VL-8B-Instruct-Q4_K_M.gguf -p "a lovely cat holding a sign says 'qwen2.1.cpp'" --cfg-scale 6.0 --sampling-method euler -v --offload-to-cpu -o qwen_image_2.1.png
```
<img alt="Qwen Image 2.1 example" src="../assets/qwen/qwen_image_2.1.png" />
To use GGUF diffusion weights, set `--diffusion-model` to the path of a file such as `qwen_image_2.1-Q4_K.gguf`.
### Image editing
Pass the reference image with `-r` and describe the edit in `-p`. Vision weights are required; the example below loads them separately with `--llm_vision`.
```powershell
.\bin\Release\sd-cli.exe --diffusion-model ..\models\diffusion_models\qwen_image_2.1_int8_convrot.safetensors --vae ..\models\vae\qwen_image_2.1_vae_bf16.safetensors --llm ..\models\text_encoders\Qwen3VL-8B-Instruct-Q4_K_M.gguf --llm_vision ..\models\text_encoders\Qwen3VL-8B-Instruct-mmproj-BF16.gguf -r ..\assets\qwen\qwen_image_2.1.png -p "change 'qwen2.1.cpp' to 'sd.cpp'" --cfg-scale 6.0 --sampling-method euler -v --offload-to-cpu -o qwen_image_2.1_edit.png
```
For multiple reference images, repeat `-r` in the desired order, for example `-r first.png -r second.png`.
+7 -7
View File
@@ -57,7 +57,7 @@ The RPC server acts as the worker. You must explicitly enable the **backend** (t
To find the correct flags for your system, refer to the official documentation for the [`llama.cpp`](https://github.com/ggml-org/llama.cpp/blob/master/docs/build.md) repository.
> **Crucial:** You must include the compiler flags required to satisfy the API compatibility with `stable-diffusion.cpp` (`-DGGML_MAX_NAME=128`). Without this flag, `GGML_MAX_NAME` will default to `64` for the server, and data transfers between the client and server will fail. Of course, `-DGGML_RPC` must also be enabled.
> **Crucial:** You must include the compiler flags required to satisfy the API compatibility with `stable-diffusion.cpp` (`-DGGML_MAX_NAME=160`). Without this flag, `GGML_MAX_NAME` will default to `64` for the server, and data transfers between the client and server will fail. Of course, `-DGGML_RPC` must also be enabled.
>
> I recommend disabling the `LLAMA_CURL` flag to avoid unnecessary dependencies, and disabling shared library builds to avoid potential conflicts.
@@ -72,8 +72,8 @@ cmake .. -DGGML_RPC=ON \
-DGGML_VULKAN=ON \ # Ensure backend is enabled
-DGGML_BUILD_SHARED_LIBS=OFF \
-DLLAMA_CURL=OFF \
-DCMAKE_C_FLAGS=-DGGML_MAX_NAME=128 \
-DCMAKE_CXX_FLAGS=-DGGML_MAX_NAME=128
-DCMAKE_C_FLAGS=-DGGML_MAX_NAME=160 \
-DCMAKE_CXX_FLAGS=-DGGML_MAX_NAME=160
cmake --build . --config Release --target rpc-server -j $(nproc)
```
@@ -86,8 +86,8 @@ cmake .. -DGGML_RPC=ON \
-DGGML_METAL=ON \
-DGGML_BUILD_SHARED_LIBS=OFF \
-DLLAMA_CURL=OFF \
-DCMAKE_C_FLAGS=-DGGML_MAX_NAME=128 \
-DCMAKE_CXX_FLAGS=-DGGML_MAX_NAME=128
-DCMAKE_C_FLAGS=-DGGML_MAX_NAME=160 \
-DCMAKE_CXX_FLAGS=-DGGML_MAX_NAME=160
cmake --build . --config Release --target rpc-server
```
@@ -101,8 +101,8 @@ cmake .. -G "Visual Studio 17 2022" -A x64 `
-DGGML_VULKAN=ON `
-DGGML_BUILD_SHARED_LIBS=OFF `
-DLLAMA_CURL=OFF `
-DCMAKE_C_FLAGS=-DGGML_MAX_NAME=128 `
-DCMAKE_CXX_FLAGS=-DGGML_MAX_NAME=128
-DCMAKE_C_FLAGS=-DGGML_MAX_NAME=160 `
-DCMAKE_CXX_FLAGS=-DGGML_MAX_NAME=160
cmake --build . --config Release --target rpc-server
```
+46
View File
@@ -0,0 +1,46 @@
# How to Use
SenseNova U1.5 is an 8B MoT model that performs diffusion directly in RGB pixel
space. It does not require a separate text encoder or VAE.
## Download weights
- Download SenseNova U1.5 8B MoT
- safetensors: https://huggingface.co/sensenova/SenseNova-U1.5-8B-MoT
Pass the complete downloaded repository directory to `--model`. The directory
must contain `model.safetensors.index.json`, every referenced Safetensors shard,
and the tokenizer files.
## Examples
### CUDA
```bash
./bin/sd-cli \
--model /path/to/SenseNova-U1.5-8B-MoT \
--prompt "a red cube on a white background" \
--width 2048 \
--height 2048 \
--steps 50 \
--cfg-scale 4 \
--flow-shift 3 \
--seed 42 \
--sampling-method euler \
--rng cuda \
--fa \
--output output.png
```
## Notes
- To match the official non-thinking text-to-image pipeline, use 50 Euler
steps, CFG 4, flow shift 3, seed 42, CUDA RNG, and an empty negative prompt.
- Width and height must be multiples of 32. The trained 1:1 resolution is
2048x2048; lower resolutions are useful for smoke tests but are outside the
training buckets.
- The SenseNova prompt template and unconditional prompt are built
automatically.
- This implementation supports non-thinking text-to-image generation. Image
editing, visual understanding, interleaved generation, and thinking-mode
prompt expansion are not implemented.
+107
View File
@@ -0,0 +1,107 @@
# JSON tokenizers
Use a Hugging Face `tokenizer.json` to supply the tokenizer vocabulary, merges,
added tokens, and processing stages. **PiD (including PiD 1.5) and Lens (including
Lens Turbo) require an external JSON**; their Gemma 2 and GPT-OSS tokenizers are
not embedded. Initialization fails if the main tokenizer is missing. Other
models keep their embedded tokenizer when this option is omitted.
```shell
sd-cli --diffusion-model model.gguf --llm text_encoder.gguf \
--tokenizer tokenizer_gemma2.json --vae vae.safetensors -p "a cat"
```
Choose the JSON belonging to the text encoder checkpoint. Checking that IDs fit
the embedding table does not establish that two vocabularies have the same
meaning. The JSON file is loaded when the text encoder is created; its embedded
vocabulary is not loaded in this case.
| Model | Required text encoder tokenizer | Example |
| --- | --- | --- |
| PiD / PiD 1.5 | Gemma 2 matching the text encoder checkpoint | `--tokenizer tokenizer_gemma2.json` |
| Lens / Lens Turbo | GPT-OSS matching the text encoder checkpoint | `--tokenizer tokenizer_gpt_oss.json` |
The Gemma 3/4 tokenizer used by LTX-2 remains embedded.
| Option | Encoder |
| --- | --- |
| `--tokenizer FILE` | Main LLM/BPE encoder: Gemma 2, Gemma 3, Qwen 2/3, Mistral, GPT-OSS; also Anima and HiDream-O1 |
| `--tokenizer FILE` | Shared CLIP tokenizer in SD1/SD2/SDXL, or CLIP-L in Flux |
| `--tokenizer clip-l=FILE` | Separate CLIP-L in SD3 or Flux |
| `--tokenizer clip-g=FILE` | Separate CLIP-G in SD3 |
Use comma-separated assignments to configure multiple slots, for example
`--tokenizer main=main.json,clip-l=clip_l.json,clip-g=clip_g.json`.
A plain file path is equivalent to `main=FILE`. You may also repeat `--tokenizer`
with explicit assignments, such as `--tokenizer main=main.json --tokenizer clip-l=clip.json`.
Empty assignment paths, unknown keys, malformed assignments and
duplicate slots are rejected. Commas separate entries in the assignment form;
quote the complete argument when paths contain spaces.
SD3 overrides must name the `clip-l` or `clip-g` slot. SDXL uses one shared
tokenizer for both CLIP encoders. Do not supply both `main` and `clip-l` for Flux.
A slot targeting an absent or unsupported encoder fails initialization.
T5/SentencePiece Unigram tokenizers are outside this implementation's scope.
For example, SD3 can load the same CLIP JSON into both slots:
```shell
sd-cli --diffusion-model sd3.gguf --clip_l clip_l.safetensors \
--clip_g clip_g.safetensors --t5xxl t5xxl.gguf --vae vae.safetensors \
--tokenizer clip-l=tokenizer_clip.json,clip-g=tokenizer_clip.json \
-p "a cat"
```
The C API accepts the same string in `sd_ctx_params_t::tokenizer`. A null or
empty value keeps an embedded tokenizer where available; PiD and Lens require
a nonempty main tokenizer path. The CLI passes the string through;
`TokenizerConfig` parses and validates it when text encoders are initialized.
```c
sd_ctx_params_t params;
sd_ctx_params_init(&params);
params.tokenizer = "clip-l=tokenizer_clip.json,clip-g=tokenizer_clip.json";
```
Rebuild applications against the updated public header when using the updated
library.
## Supported components
| Stage | Supported configurations |
| --- | --- |
| Normalizer | `Sequence`, `NFC`, `Lowercase`, `Replace` with String/Regex patterns |
| PreTokenizer | `Sequence`, `Split` with String/Regex patterns, all five delimiter behaviors and `invert`; `ByteLevel` with `add_prefix_space` and `use_regex` |
| Model | Deterministic `BPE`, string or array-pair merges, `unk_token`, `fuse_unk`, `byte_fallback`, `ignore_merges`, `end_of_word_suffix` |
| PostProcessor | Single-sequence `TemplateProcessing` with at most one prefix and one suffix token, `RobertaProcessing`, `ByteLevel` |
| Decoder | `Sequence`, `Replace`, `ByteLevel`, `ByteFallback`, `Fuse` |
| AddedToken | Special and ordinary added tokens, original IDs, raw or normalized matching, leftmost-longest matching |
`ByteLevel.use_regex` defaults to true when omitted. ByteLevel postprocessing
changes offsets only and adds no tokens. Added tokens with `single_word`,
`lstrip`, or `rstrip` enabled, nonzero BPE dropout, nonempty
`continuing_subword_prefix`, and unsupported component types fail loading.
New added-token IDs must follow the model vocabulary consecutively; configurations
whose IDs Hugging Face would reassign are rejected.
JSON `padding` and `truncation` must be null. This API returns IDs, not offsets,
type IDs, or paired-input encodings; the pair template is not used.
The pipeline covers the CLIP, Gemma 2, Gemma 3, GPT-OSS, Mistral 3, Qwen 2 and
Qwen 3 JSON configurations used by the differential test. It does not imply
support for every tokenizer published under those model names.
## Prompt integration
Prompt attention parsing and model-specific chat/image templates remain in the
conditioner. Raw `encode()` does not add BOS/EOS. The conditioner concatenates
weighted prompt fragments, then the existing padding/chunking step applies the
JSON single-sequence template once per sequence or CLIP chunk. Padding ID,
direction, length limits and attention masks remain text encoder policies.
CLIP requires both BOS and EOS because its chunking reserves those positions.
The internal `encode()`, `tokenize()`, and `decode()` interfaces return a success
flag and write to an output parameter. A successful result may be empty; a failed
call clears its output. JSON tokenizer input, normalization, and regex failures
return `false` with diagnostic information instead of throwing. Invalid
JSON, unsupported stages, conflicting IDs and IDs outside the encoder embedding
table fail initialization instead of falling back to the embedded tokenizer.
+45
View File
@@ -0,0 +1,45 @@
# Troubleshooting
## Completely black or white images or videos / NaNs
Some ggml backends can encounter numerical overflow during inference, producing
NaN (not-a-number) values. This can result in completely black or white images or videos.
Whether it happens can depend on the backend, device, model, and weight format.
Known overflow issues have been addressed as far as possible, but the maintainer
has limited hardware and cannot test every combination. Some cases may therefore
still need a manual workaround.
These options are supported by both `sd-cli` and `sd-server`. If you encounter
this problem, add them to your CLI generation command or server startup command:
```sh
--linear-scale 0.0078125 --attn-scale 0.0078125
```
For `sd-server`, restart the server after changing these startup options. Run the
same prompt and seed again to see whether the output recovers. If the problem
persists, try smaller positive values, for example:
```sh
--linear-scale 0.00390625 --attn-scale 0.00390625
```
These options reduce intermediate values and compensate afterwards to preserve
the intended output scale:
- `--linear-scale` scales Linear inputs before matrix multiplication and rescales
the result.
- `--attn-scale` scales attention keys and values (K/V). It takes effect only in
the Flash Attention path, where `--fa` or `--diffusion-fa` is enabled and the
backend supports it.
The two values can be set independently and apply across model components. The
default `0` preserves each model's built-in settings; `1` explicitly disables the
corresponding scaling. Overrides must be finite positive values. C API users can
set `linear_scale` and `attn_scale` in `sd_ctx_params_t`.
If the problem persists after trying the relevant steps above,
[submit a bug report](https://github.com/leejet/stable-diffusion.cpp/issues/new?template=bug_report.yml).
Include your full command, backend and hardware, model and weight format, logs,
and the scale values you tried with their results.
+49
View File
@@ -34,6 +34,10 @@
- Wan2.2 I2V A14B
- safetensors: https://huggingface.co/Comfy-Org/Wan_2.2_ComfyUI_Repackaged/tree/main/split_files/diffusion_models
- gguf: https://huggingface.co/QuantStack/Wan2.2-I2V-A14B-GGUF/tree/main
- Wan2.2 S2V 14B
- safetensors: https://huggingface.co/Comfy-Org/Wan_2.2_ComfyUI_Repackaged/tree/main/split_files/diffusion_models
- gguf: https://huggingface.co/QuantStack/Wan2.2-S2V-14B-GGUF/tree/main
- int8_convrot safetensors: https://huggingface.co/noctrex/Wan2.2-S2V-14B-int8_convrot
- Download vae
- wan_2.1_vae (for all the wan model except Wan2.2 TI2V 5B)
- safetensors: https://huggingface.co/Comfy-Org/Wan_2.1_ComfyUI_repackaged/blob/main/split_files/vae/wan_2.1_vae.safetensors
@@ -49,6 +53,9 @@
- Download clip_vison_h (for Wan2.1 I2V/FLF2V only)
- safetensors: https://huggingface.co/Comfy-Org/Wan_2.1_ComfyUI_repackaged/blob/main/split_files/clip_vision/clip_vision_h.safetensors
- Download audio_encoder (for Wan2.2 S2V only)
- safetensors: https://huggingface.co/Comfy-Org/Wan_2.2_ComfyUI_Repackaged/blob/main/split_files/audio_encoders/wav2vec2_large_english_fp16.safetensors
## Examples
@@ -94,6 +101,48 @@
<video src=../assets/wan/Wan2.2_14B_i2v.mp4 controls="controls" muted="muted" type="video/mp4"></video>
### Wan2.2 S2V 14B
Audio-driven video (speech-to-video). The reference image (`-i`) is the speaker
portrait, `--audio` is the driving audio track and `--audio-encoder` is the
wav2vec2 audio encoder. Wan2.2 S2V requires the wan_2.1 vae (16 channel), not
the wan2.2 vae.
```
.\bin\Release\sd-cli.exe -M vid_gen --diffusion-model ..\models\diffusion_models\wan2.2_s2v-14B-Q8_0.gguf --audio-encoder ..\models\audio_encoders\wav2vec2_large_english_fp16.safetensors --vae ..\models\vae\wan_2.1_vae.safetensors --t5xxl ..\models\text_encoders\umt5-xxl-encoder-Q8_0.gguf -p "a person is talking" --cfg-scale 6.0 --steps 20 --sampling-method euler -v -n "色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走" -W 832 -H 480 --diffusion-fa --offload-to-cpu --vae-tiling --video-frames 81 -i ..\assets\cat_with_sd_cpp_42.png --audio .\input\speech.wav --flow-shift 3.0
```
Notes:
- Recommended settings: `--sampling-method euler --steps 20 --cfg-scale 6.0`.
`dpm++2m` produces heavy artifacts on S2V. 4 steps with the lightning LoRA
(below) is the fast option.
- Resolutions: width and height must be multiples of 16; the examples use
multiples of 64. 832x480 is a fast starting point; generation cost scales
with pixel area.
- `--audio` accepts a WAV file; it is downmixed to mono and resampled to 16 kHz
internally. Audio longer than the video is truncated, video longer than the
audio is padded with silence. Pick `--video-frames` to match the audio:
roughly `audio_seconds * 16` frames, capped at one chunk (77-81 frames,
~5 s at the model's 16 fps). 33, 77 and 81 map to clean latent frame counts.
- S2V always uses 16 fps. Other requested frame rates are automatically
changed to 16 with a warning, including the CLI and server video output.
`generate_video()` returns the actual frame rate through `fps_out`; C API
callers should use that value when encoding the output video.
- One generation covers the first S2V chunk window (`--video-frames` frames).
Long-video chunked extend mode is not implemented yet.
- Speed: the lightx2v lightning LoRA works with S2V at 4 steps and
`--cfg-scale 1.0`. Use the **low_noise** variant;
the high_noise variant produces artifacts on S2V:
```
--lora-model-dir ..\models\loras
-p "...<lora:lightx2v-Wan2.2-T2V-A14B-4steps-lora-rank64-Seko-V2.0-low_noise:1.0>"
--cfg-scale 1.0 --steps 4
```
Expect some quality/dynamics loss compared to the full 20-step run.
### Wan2.2 T2V A14B T2I
```
+3
View File
@@ -22,3 +22,6 @@ Metadata mode inspects PNG/JPEG container metadata without loading any model:
./bin/sd-cli -M metadata --image ./output.png --metadata-raw
./bin/sd-cli -M metadata --image ./output.png --metadata-all
```
For completely black or white images or videos, NaNs, and the `--linear-scale` /
`--attn-scale` workaround, see [Troubleshooting](../../docs/troubleshooting.md).
+9 -5
View File
@@ -419,7 +419,8 @@ void step_callback(int step, int frame_count, sd_image_t* image, bool is_noisy,
LOG_ERROR("save preview image to '%s' failed", path.string().c_str());
}
} else {
if (create_video_from_sd_images(cli_params->preview_path.c_str(), image, frame_count, cli_params->preview_fps, cli_params->compression_quality) != 0) {
int fps = cli_params->preview_method == PREVIEW_PROJ ? cli_params->preview_fps / 4 : cli_params->preview_fps;
if (create_video_from_sd_images(cli_params->preview_path.c_str(), image, frame_count, fps, cli_params->compression_quality) != 0) {
LOG_ERROR("save preview video to '%s' failed", cli_params->preview_path.c_str());
}
}
@@ -540,12 +541,16 @@ bool save_results(const SDCliParams& cli_params,
if (cli_params.mode == VID_GEN && num_results > 1) {
if (ext_lower != ".avi" && ext_lower != ".webp" && ext_lower != ".webm")
ext = ".avi";
std::string params = gen_params.embed_image_metadata
? get_image_params(ctx_params, gen_params, gen_params.seed, cli_params.mode)
: "";
fs::path video_path = base_path;
video_path += ext;
std::string final_ext_lower = ext.string();
std::transform(final_ext_lower.begin(), final_ext_lower.end(), final_ext_lower.begin(), ::tolower);
const bool mux_audio = generated_audio != nullptr && (final_ext_lower == ".avi" || final_ext_lower == ".webm");
if (create_video_from_sd_images(video_path.string().c_str(), results, num_results, gen_params.fps, cli_params.compression_quality, mux_audio ? generated_audio : nullptr) == 0) {
if (create_video_from_sd_images(video_path.string().c_str(), results, num_results, gen_params.fps, cli_params.compression_quality, mux_audio ? generated_audio : nullptr, params) == 0) {
LOG_INFO("save result video to '%s'", video_path.string().c_str());
if (generated_audio != nullptr && !mux_audio) {
fs::path wav_path = video_path;
@@ -687,8 +692,6 @@ int main(int argc, const char* argv[]) {
}
}
cli_params.preview_fps = gen_params.fps;
if (cli_params.preview_method == PREVIEW_PROJ)
cli_params.preview_fps /= 4;
sd_set_preview_callback(step_callback,
cli_params.preview_method,
@@ -951,9 +954,10 @@ int main(int argc, const char* argv[]) {
} else if (cli_params.mode == VID_GEN) {
sd_vid_gen_params_t vid_gen_params = gen_params.to_sd_vid_gen_params_t();
sd_image_t* generated_video = nullptr;
if (!generate_video(sd_ctx.get(), &vid_gen_params, &generated_video, &num_results, &generated_audio)) {
if (!generate_video(sd_ctx.get(), &vid_gen_params, &generated_video, &num_results, &generated_audio, &cli_params.preview_fps)) {
generated_video = nullptr;
}
gen_params.fps = cli_params.preview_fps;
results.adopt(generated_video, num_results);
}
+79 -9
View File
@@ -302,8 +302,12 @@ bool parse_options(int argc, const char** argv, const std::vector<ArgOptions>& o
invalid_arg = true;
return;
}
*option.target = std::stoi(argv[i]);
found_arg = true;
try {
*option.target = std::stoi(argv[i]);
} catch (const std::invalid_argument&) {
invalid_arg = true;
}
found_arg = true;
}))
break;
@@ -312,8 +316,12 @@ bool parse_options(int argc, const char** argv, const std::vector<ArgOptions>& o
invalid_arg = true;
return;
}
*option.target = std::stof(argv[i]);
found_arg = true;
try {
*option.target = std::stof(argv[i]);
} catch (const std::invalid_argument&) {
invalid_arg = true;
}
found_arg = true;
}))
break;
@@ -337,7 +345,8 @@ bool parse_options(int argc, const char** argv, const std::vector<ArgOptions>& o
if (invalid_arg) {
if (!valid) {
LOG_ERROR("error: invalid parameter for argument: %s", arg.c_str());
LOG_ERROR("error: invalid parameter for argument \"%s\": \"%s\"",
arg.c_str(), (i >= argc) ? "" : argv[i]);
}
return false;
}
@@ -350,6 +359,25 @@ bool parse_options(int argc, const char** argv, const std::vector<ArgOptions>& o
return true;
}
static int parse_scale_override(int argc, const char** argv, int index, float& scale) {
if (++index >= argc) {
return -1;
}
try {
size_t end = 0;
const std::string value = argv[index];
float parsed = std::stof(value, &end);
if (end != value.size() || !std::isfinite(parsed) || parsed < 0.f ||
(parsed > 0.f && !std::isfinite(1.f / parsed))) {
return -1;
}
scale = parsed;
} catch (const std::exception&) {
return -1;
}
return 1;
}
ArgOptions SDContextParams::get_options() {
ArgOptions options;
options.string_options = {
@@ -382,6 +410,11 @@ ArgOptions SDContextParams::get_options() {
"path to the llm text encoder. For example: (qwenvl2.5 for qwen-image, mistral-small3.2 for flux2, ...)",
0,
&llm_path},
{"",
"--tokenizer",
"tokenizer.json path, or comma-separated main=FILE,clip-l=FILE,clip-g=FILE assignments; required for PiD and Lens",
(int)',',
&tokenizer},
{"",
"--llm_vision",
"path to the llm vit",
@@ -432,6 +465,11 @@ ArgOptions SDContextParams::get_options() {
"path to standalone LTX audio vae model",
0,
&audio_vae_path},
{"",
"--audio-encoder",
"path to wav2vec2 audio encoder model (Wan2.2 S2V)",
0,
&audio_encoder_path},
{"",
"--taesd",
"path to taesd. Using Tiny AutoEncoder for fast decoding (low quality)",
@@ -586,7 +624,7 @@ ArgOptions SDContextParams::get_options() {
true, &diffusion_conv_direct},
{"",
"--vae-conv-direct",
"use ggml_conv2d_direct in the vae model",
"use direct 2D and 3D convolutions in the vae model",
true, &vae_conv_direct},
};
@@ -678,11 +716,23 @@ ArgOptions SDContextParams::get_options() {
};
options.manual_options = {
{"",
"--linear-scale",
"linear input scale override (float, default: 0 = model default, 1 = no scaling)",
[this](int argc, const char** argv, int index) {
return parse_scale_override(argc, argv, index, linear_scale);
}},
{"",
"--attn-scale",
"flash-attention K/V scale override (float, default: 0 = model default, 1 = no scaling); requires --fa or --diffusion-fa",
[this](int argc, const char** argv, int index) {
return parse_scale_override(argc, argv, index, attn_scale);
}},
{"",
"--auto-fit",
"on|off (default: on). Use one GPU for diffusion/te/vae computation and place weights on that GPU, "
"on|off (default: on). Preserve --backend (otherwise select one GPU) and place weights on the compute GPU, "
"RAM, another GPU, or disk in that order, according to available memory (--max-vram limits GPU budgets). "
"Disabled by explicit --backend or --params-backend; uses automatic graph segmentation when needed",
"Disabled by explicit --params-backend; uses automatic graph segmentation when needed",
on_auto_fit_arg},
{"",
"--type",
@@ -851,6 +901,7 @@ std::string SDContextParams::to_string() const {
<< " t5xxl_path: \"" << t5xxl_path << "\",\n"
<< " llm_path: \"" << llm_path << "\",\n"
<< " llm_vision_path: \"" << llm_vision_path << "\",\n"
<< " tokenizer: \"" << tokenizer << "\",\n"
<< " diffusion_model_path: \"" << diffusion_model_path << "\",\n"
<< " high_noise_diffusion_model_path: \"" << high_noise_diffusion_model_path << "\",\n"
<< " uncond_diffusion_model_path: \"" << uncond_diffusion_model_path << "\",\n"
@@ -858,6 +909,7 @@ std::string SDContextParams::to_string() const {
<< " vae_path: \"" << vae_path << "\",\n"
<< " vae_format: \"" << vae_format << "\",\n"
<< " audio_vae_path: \"" << audio_vae_path << "\",\n"
<< " audio_encoder_path: \"" << audio_encoder_path << "\",\n"
<< " taesd_path: \"" << taesd_path << "\",\n"
<< " esrgan_path: \"" << esrgan_path << "\",\n"
<< " control_net_path: \"" << control_net_path << "\",\n"
@@ -886,6 +938,8 @@ std::string SDContextParams::to_string() const {
<< " vae_on_cpu: " << (vae_on_cpu ? "true" : "false") << ",\n"
<< " flash_attn: " << (flash_attn ? "true" : "false") << ",\n"
<< " diffusion_flash_attn: " << (diffusion_flash_attn ? "true" : "false") << ",\n"
<< " linear_scale: " << linear_scale << ",\n"
<< " attn_scale: " << attn_scale << ",\n"
<< " diffusion_conv_direct: " << (diffusion_conv_direct ? "true" : "false") << ",\n"
<< " vae_conv_direct: " << (vae_conv_direct ? "true" : "false") << ",\n"
<< " prediction: " << sd_prediction_name(prediction) << ",\n"
@@ -915,12 +969,14 @@ sd_ctx_params_t SDContextParams::to_sd_ctx_params_t(bool taesd_preview) {
sd_ctx_params.t5xxl_path = t5xxl_path.c_str();
sd_ctx_params.llm_path = llm_path.c_str();
sd_ctx_params.llm_vision_path = llm_vision_path.c_str();
sd_ctx_params.tokenizer = tokenizer.c_str();
sd_ctx_params.diffusion_model_path = diffusion_model_path.c_str();
sd_ctx_params.high_noise_diffusion_model_path = high_noise_diffusion_model_path.c_str();
sd_ctx_params.uncond_diffusion_model_path = uncond_diffusion_model_path.c_str();
sd_ctx_params.embeddings_connectors_path = embeddings_connectors_path.c_str();
sd_ctx_params.vae_path = vae_path.c_str();
sd_ctx_params.audio_vae_path = audio_vae_path.c_str();
sd_ctx_params.audio_encoder_path = audio_encoder_path.c_str();
sd_ctx_params.taesd_path = taesd_path.c_str();
sd_ctx_params.control_net_path = control_net_path.c_str();
sd_ctx_params.ip_adapter_path = ip_adapter_path.c_str();
@@ -939,6 +995,8 @@ sd_ctx_params_t SDContextParams::to_sd_ctx_params_t(bool taesd_preview) {
sd_ctx_params.enable_mmap = enable_mmap;
sd_ctx_params.flash_attn = flash_attn;
sd_ctx_params.diffusion_flash_attn = diffusion_flash_attn;
sd_ctx_params.linear_scale = linear_scale;
sd_ctx_params.attn_scale = attn_scale;
sd_ctx_params.tae_preview_only = taesd_preview;
sd_ctx_params.diffusion_conv_direct = diffusion_conv_direct;
sd_ctx_params.vae_conv_direct = vae_conv_direct;
@@ -1051,7 +1109,7 @@ ArgOptions SDGenerationParams::get_options() {
&hires_upscaler},
{"",
"--extra-sample-args",
"extra sampler/scheduler/guidance args, key=value list. CFG supports guidance_schedule; APG supports apg_eta, apg_momentum, apg_norm_threshold, apg_norm_threshold_smoothing; SLG supports slg_uncond; lcm supports noise_clip_std, noise_scale_start, noise_scale_end; flux supports base_shift, max_shift; ltx2 supports max_shift, base_shift, stretch, terminal; euler_ge supports gamma; beta scheduler supports alpha, beta; logit_normal supports mu, std, logsnr_min, logsnr_max, resolution_aware; lms supports lms_max_order, lms_shift, lms_divisions",
"extra sampler/scheduler/guidance args, key=value list. CFG supports guidance_schedule; APG supports apg_eta, apg_momentum, apg_norm_threshold, apg_norm_threshold_smoothing; SLG supports slg_uncond; lcm supports noise_clip_std, noise_scale_start, noise_scale_end; flux supports base_shift, max_shift; ltx2 supports max_shift, base_shift, stretch, terminal; euler_ge supports gamma; beta scheduler supports alpha, beta; logit_normal supports mu, std, logsnr_min, logsnr_max, resolution_aware; lms supports lms_max_order, lms_shift, lms_divisions; noise-injecting samplers support noise_sampler with value iid (default except for dpm++2m_sde_bt) or brownian_tree; brownian_tree_rng supports cpu (default), cuda, std_default or sampler_rng",
(int)',',
&extra_sample_args},
{"",
@@ -1471,6 +1529,14 @@ ArgOptions SDGenerationParams::get_options() {
return 1;
};
auto on_audio_arg = [&](int argc, const char** argv, int index) {
if (++index >= argc) {
return -1;
}
ref_audio_paths.push_back(argv[index]);
return 1;
};
auto on_cache_mode_arg = [&](int argc, const char** argv, int index) {
if (++index >= argc) {
return -1;
@@ -1660,6 +1726,10 @@ ArgOptions SDGenerationParams::get_options() {
"--ref-audio",
"standalone WAV reference for MiniMax-H3 Ref2VA (can be used multiple times)",
on_ref_audio_arg},
{"",
"--audio",
"driving audio track (Wan2.2 S2V; can be used once)",
on_audio_arg},
{"",
"--cache-mode",
"caching method: 'easycache' (DiT), 'ucache' (UNET), 'dbcache'/'taylorseer'/'cache-dit' (DiT block-level), 'spectrum' (UNET/DiT Chebyshev+Taylor forecasting)",
+4
View File
@@ -124,6 +124,7 @@ struct SDContextParams {
std::string t5xxl_path;
std::string llm_path;
std::string llm_vision_path;
std::string tokenizer;
std::string diffusion_model_path;
std::string high_noise_diffusion_model_path;
std::string uncond_diffusion_model_path;
@@ -131,6 +132,7 @@ struct SDContextParams {
std::string vae_path;
std::string vae_format = "auto";
std::string audio_vae_path;
std::string audio_encoder_path;
std::string taesd_path;
std::string esrgan_path;
std::string control_net_path;
@@ -175,6 +177,8 @@ struct SDContextParams {
lora_apply_mode_t lora_apply_mode = LORA_APPLY_AUTO;
bool force_sdxl_vae_conv_scale = false;
float linear_scale = 0.f;
float attn_scale = 0.f;
float flow_shift = INFINITY;
ArgOptions get_options();
+53 -11
View File
@@ -810,7 +810,31 @@ uint8_t* load_image_from_memory(const char* image_bytes,
return load_image_common(true, image_bytes, len, width, height, expected_width, expected_height, expected_channel);
}
std::vector<uint8_t> create_mjpg_avi_from_sd_images_to_vector(sd_image_t* images, int num_images, int fps, int quality, const sd_audio_t* audio) {
static void append_avi_metadata(std::vector<uint8_t>& data, const std::string& parameters) {
if (parameters.empty()) {
return;
}
std::vector<uint8_t> info_content;
write_fourcc(info_content, "INFO");
const size_t comment_size = parameters.size() + 1;
write_fourcc(info_content, "ICMT");
write_u32_le(info_content, static_cast<uint32_t>(comment_size));
info_content.insert(info_content.end(), parameters.begin(), parameters.end());
info_content.push_back(0);
if (comment_size & 1u) {
info_content.push_back(0);
}
write_fourcc(data, "LIST");
write_u32_le(data, static_cast<uint32_t>(info_content.size()));
data.insert(data.end(), info_content.begin(), info_content.end());
size_t start_pos = data.size();
}
std::vector<uint8_t> create_mjpg_avi_from_sd_images_to_vector(sd_image_t* images, int num_images, int fps, int quality, const sd_audio_t* audio, const std::string& parameters) {
if (num_images == 0) {
fprintf(stderr, "Error: Image array is empty.\n");
return {};
@@ -1000,6 +1024,8 @@ std::vector<uint8_t> create_mjpg_avi_from_sd_images_to_vector(sd_image_t* images
const size_t movi_size = avi_data.size() - movi_size_pos - 4;
patch_u32_le(avi_data, movi_size_pos, static_cast<uint32_t>(movi_size));
append_avi_metadata(avi_data, parameters);
write_fourcc(avi_data, "idx1");
write_u32_le(avi_data, static_cast<uint32_t>(index.size() * 16));
for (const auto& entry : index) {
@@ -1015,8 +1041,8 @@ std::vector<uint8_t> create_mjpg_avi_from_sd_images_to_vector(sd_image_t* images
return avi_data;
}
int create_mjpg_avi_from_sd_images(const char* filename, sd_image_t* images, int num_images, int fps, int quality, const sd_audio_t* audio) {
std::vector<uint8_t> avi_data = create_mjpg_avi_from_sd_images_to_vector(images, num_images, fps, quality, audio);
int create_mjpg_avi_from_sd_images(const char* filename, sd_image_t* images, int num_images, int fps, int quality, const sd_audio_t* audio, const std::string& parameters) {
std::vector<uint8_t> avi_data = create_mjpg_avi_from_sd_images_to_vector(images, num_images, fps, quality, audio, parameters);
if (avi_data.empty()) {
return -1;
}
@@ -1146,7 +1172,7 @@ int create_animated_webp_from_sd_images(const char* filename, sd_image_t* images
#endif
#ifdef SD_USE_WEBM
std::vector<uint8_t> create_webm_from_sd_images_to_vector(sd_image_t* images, int num_images, int fps, int quality, const sd_audio_t* audio) {
std::vector<uint8_t> create_webm_from_sd_images_to_vector(sd_image_t* images, int num_images, int fps, int quality, const sd_audio_t* audio, const std::string& parameters) {
if (num_images == 0) {
fprintf(stderr, "Error: Image array is empty.\n");
return {};
@@ -1213,6 +1239,21 @@ std::vector<uint8_t> create_webm_from_sd_images_to_vector(sd_image_t* images, in
segment.GetSegmentInfo()->set_writing_app("stable-diffusion.cpp");
segment.GetSegmentInfo()->set_muxing_app("stable-diffusion.cpp");
LOG_DEBUG("Embedding parameters to metadata: %s", parameters.c_str());
if (!parameters.empty()) {
mkvmuxer::Tag* tag = segment.AddTag();
if (tag) {
if (!tag->add_simple_tag("COMMENT", parameters.c_str())) {
LOG_WARN("Failed to add COMMENT simple tag.");
}
} else {
LOG_WARN("Failed to add tag to segment.");
}
} else {
LOG_INFO("Paramaters is empty, COMMENT tag not embedded.\n");
}
const uint64_t frame_duration_ns = std::max<uint64_t>(
1, static_cast<uint64_t>(std::llround(1000000000.0 / static_cast<double>(fps))));
uint64_t timestamp_ns = 0;
@@ -1271,8 +1312,8 @@ std::vector<uint8_t> create_webm_from_sd_images_to_vector(sd_image_t* images, in
return writer.data();
}
int create_webm_from_sd_images(const char* filename, sd_image_t* images, int num_images, int fps, int quality, const sd_audio_t* audio) {
std::vector<uint8_t> webm_data = create_webm_from_sd_images_to_vector(images, num_images, fps, quality, audio);
int create_webm_from_sd_images(const char* filename, sd_image_t* images, int num_images, int fps, int quality, const sd_audio_t* audio, const std::string& parameters) {
std::vector<uint8_t> webm_data = create_webm_from_sd_images_to_vector(images, num_images, fps, quality, audio, parameters);
if (webm_data.empty()) {
return -1;
}
@@ -1289,7 +1330,8 @@ std::vector<uint8_t> create_video_from_sd_images_to_vector(const std::string& ou
int num_images,
int fps,
int quality,
const sd_audio_t* audio) {
const sd_audio_t* audio,
const std::string& parameters) {
std::string format = output_format;
std::transform(format.begin(), format.end(), format.begin(),
[](unsigned char c) { return static_cast<char>(tolower(c)); });
@@ -1299,7 +1341,7 @@ std::vector<uint8_t> create_video_from_sd_images_to_vector(const std::string& ou
#ifdef SD_USE_WEBM
if (format == "webm") {
return create_webm_from_sd_images_to_vector(images, num_images, fps, quality, audio);
return create_webm_from_sd_images_to_vector(images, num_images, fps, quality, audio, parameters);
}
#endif
@@ -1309,14 +1351,14 @@ std::vector<uint8_t> create_video_from_sd_images_to_vector(const std::string& ou
}
#endif
return create_mjpg_avi_from_sd_images_to_vector(images, num_images, fps, quality, audio);
return create_mjpg_avi_from_sd_images_to_vector(images, num_images, fps, quality, audio, parameters);
}
int create_video_from_sd_images(const char* filename, sd_image_t* images, int num_images, int fps, int quality, const sd_audio_t* audio) {
int create_video_from_sd_images(const char* filename, sd_image_t* images, int num_images, int fps, int quality, const sd_audio_t* audio, const std::string& parameters) {
std::string path = filename ? filename : "";
auto pos = path.find_last_of('.');
std::string ext = pos == std::string::npos ? "" : path.substr(pos);
std::vector<uint8_t> video_data = create_video_from_sd_images_to_vector(ext, images, num_images, fps, quality, audio);
std::vector<uint8_t> video_data = create_video_from_sd_images_to_vector(ext, images, num_images, fps, quality, audio, parameters);
if (video_data.empty()) {
return -1;
}
+18 -12
View File
@@ -57,13 +57,15 @@ int create_mjpg_avi_from_sd_images(const char* filename,
sd_image_t* images,
int num_images,
int fps,
int quality = 90,
const sd_audio_t* audio = nullptr);
int quality = 90,
const sd_audio_t* audio = nullptr,
const std::string& parameters = "");
std::vector<uint8_t> create_mjpg_avi_from_sd_images_to_vector(sd_image_t* images,
int num_images,
int fps,
int quality = 90,
const sd_audio_t* audio = nullptr);
int quality = 90,
const sd_audio_t* audio = nullptr,
const std::string& parameters = "");
#ifdef SD_USE_WEBP
int create_animated_webp_from_sd_images(const char* filename,
@@ -82,27 +84,31 @@ int create_webm_from_sd_images(const char* filename,
sd_image_t* images,
int num_images,
int fps,
int quality = 90,
const sd_audio_t* audio = nullptr);
int quality = 90,
const sd_audio_t* audio = nullptr,
const std::string& parameters = "");
std::vector<uint8_t> create_webm_from_sd_images_to_vector(sd_image_t* images,
int num_images,
int fps,
int quality = 90,
const sd_audio_t* audio = nullptr);
int quality = 90,
const sd_audio_t* audio = nullptr,
const std::string& parameters = "");
#endif
int create_video_from_sd_images(const char* filename,
sd_image_t* images,
int num_images,
int fps,
int quality = 90,
const sd_audio_t* audio = nullptr);
int quality = 90,
const sd_audio_t* audio = nullptr,
const std::string& parameters = "");
std::vector<uint8_t> create_video_from_sd_images_to_vector(const std::string& output_format,
sd_image_t* images,
int num_images,
int fps,
int quality = 90,
const sd_audio_t* audio = nullptr);
int quality = 90,
const sd_audio_t* audio = nullptr,
const std::string& parameters = "");
bool write_wav_to_file(const std::string& path,
const float* interleaved_samples,
+3
View File
@@ -129,3 +129,6 @@ For detailed command-line arguments, run:
```bash
./bin/sd-server -h
```
For completely black or white images or videos, NaNs, and the `--linear-scale` /
`--attn-scale` startup options, see [Troubleshooting](../../docs/troubleshooting.md).
+7 -4
View File
@@ -237,6 +237,9 @@ bool execute_vid_gen_job(ServerRuntime& runtime,
int& output_fps,
std::string& error_message) {
sd_vid_gen_params_t params = job.vid_gen.to_sd_vid_gen_params_t();
std::string str_params = job.vid_gen.gen_params.embed_image_metadata
? get_image_params(*runtime.ctx_params, job.vid_gen.gen_params, job.vid_gen.gen_params.seed, VID_GEN)
: "";
SDImageVec results;
int num_results = 0;
@@ -245,7 +248,7 @@ bool execute_vid_gen_job(ServerRuntime& runtime,
{
std::lock_guard<std::mutex> lock(*runtime.sd_ctx_mutex);
sd_image_t* raw_results = nullptr;
if (!generate_video(runtime.sd_ctx, &params, &raw_results, &num_results, &generated_audio)) {
if (!generate_video(runtime.sd_ctx, &params, &raw_results, &num_results, &generated_audio, &output_fps)) {
raw_results = nullptr;
}
results.adopt(raw_results, num_results);
@@ -261,9 +264,10 @@ bool execute_vid_gen_job(ServerRuntime& runtime,
std::vector<uint8_t> video_bytes = create_video_from_sd_images_to_vector(job.vid_gen.output_format,
results.data(),
num_results,
job.vid_gen.gen_params.fps,
output_fps,
job.vid_gen.output_compression,
generated_audio);
generated_audio,
str_params);
free_sd_audio(generated_audio);
if (video_bytes.empty()) {
error_message = "failed to encode generated video container";
@@ -273,7 +277,6 @@ bool execute_vid_gen_job(ServerRuntime& runtime,
output_media_b64 = base64_encode(video_bytes);
output_media_mime_type = video_mime_type(job.vid_gen.output_format);
output_frame_count = num_results;
output_fps = job.vid_gen.gen_params.fps;
return true;
}
+1 -1
Submodule ggml updated: e20c3a14aa...c6632cd905
+11 -1
View File
@@ -92,6 +92,7 @@ enum prediction_t {
FLUX_FLOW_PRED,
SEFI_FLOW_PRED,
MINIT2I_FLOW_PRED,
SENSENOVA_U1_FLOW_PRED,
PREDICTION_COUNT
};
@@ -207,6 +208,7 @@ typedef struct {
const char* embeddings_connectors_path;
const char* vae_path;
const char* audio_vae_path;
const char* audio_encoder_path;
const char* taesd_path;
const char* control_net_path;
const char* ip_adapter_path;
@@ -240,6 +242,9 @@ typedef struct {
const char* rpc_servers;
const char* model_args;
bool disable_segmented_compute; // Force monolithic graph execution even when automatic graph cutting would fit memory better
float linear_scale; // Override linear input scaling; 0 keeps the model default
float attn_scale; // Override flash-attention K/V scaling; 0 keeps the model default
const char* tokenizer; // tokenizer.json path or main=FILE,clip-l=FILE,clip-g=FILE assignments; required for PiD and Lens
} sd_ctx_params_t;
typedef struct {
@@ -493,6 +498,9 @@ SD_API void free_sd_audio(sd_audio_t* audio);
SD_API void sd_sample_params_init(sd_sample_params_t* sample_params);
SD_API char* sd_sample_params_to_str(const sd_sample_params_t* sample_params);
// Requires a loaded context; returns a static string owned by the library, or "Unknown".
SD_API const char* sd_get_model_version_name(const sd_ctx_t* sd_ctx);
SD_API enum sample_method_t sd_get_default_sample_method(const sd_ctx_t* sd_ctx);
SD_API enum scheduler_t sd_get_default_scheduler(const sd_ctx_t* sd_ctx, enum sample_method_t sample_method);
@@ -515,11 +523,13 @@ enum sd_cancel_mode_t {
SD_API void sd_cancel_generation(sd_ctx_t* sd_ctx, enum sd_cancel_mode_t mode);
SD_API void sd_vid_gen_params_init(sd_vid_gen_params_t* sd_vid_gen_params);
// If non-NULL, fps_out receives the effective encoding frame rate before preview callbacks.
SD_API bool generate_video(sd_ctx_t* sd_ctx,
const sd_vid_gen_params_t* sd_vid_gen_params,
sd_image_t** frames_out,
int* num_frames_out,
sd_audio_t** audio_out);
sd_audio_t** audio_out,
int* fps_out);
typedef struct upscaler_ctx_t upscaler_ctx_t;
+2
View File
@@ -11,6 +11,8 @@ $patterns = @(
"src/extensions/*.cpp"
"src/extensions/*.h"
"src/extensions/*.hpp"
"src/pipeline/*.cpp"
"src/pipeline/*.h"
"src/runtime/*.cpp"
"src/runtime/*.h"
"src/runtime/*.hpp"
+1
View File
@@ -9,6 +9,7 @@ for f in src/*.cpp src/*.h src/*.hpp \
src/conditioning/*.cpp src/conditioning/*.h src/conditioning/*.hpp \
src/core/*.cpp src/core/*.h src/core/*.hpp \
src/extensions/*.cpp src/extensions/*.h src/extensions/*.hpp \
src/pipeline/*.cpp src/pipeline/*.h \
src/runtime/*.cpp src/runtime/*.h src/runtime/*.hpp \
src/model/*/*.cpp src/model/*/*.h src/model/*/*.hpp \
src/tokenizers/*.h src/tokenizers/*.cpp src/tokenizers/vocab/*.h src/tokenizers/vocab/*.cpp \
+435 -82
View File
@@ -7,6 +7,7 @@
#include <limits>
#include <optional>
#include <sstream>
#include <stdexcept>
#include "core/ggml_tensor_utils.h"
#include "core/tensor_ggml.hpp"
@@ -16,6 +17,8 @@
#include "model/te/llm.hpp"
#include "model/te/t5.hpp"
#include "model_loader.h"
#include "tokenizers/sensenova_u1_tokenizer.h"
#include "tokenizers/tokenizer_config.h"
struct SDCondition {
sd::Tensor<float> c_crossattn;
@@ -149,6 +152,7 @@ public:
virtual void set_graph_cut_layer_split_backend_vram_limits(const std::vector<size_t>& limits) {}
virtual void get_layer_split_param_tensors(std::map<std::string, ggml_tensor*>& tensors) {}
virtual void set_flash_attention_enabled(bool enabled) = 0;
virtual void set_scale_overrides(float linear_scale, float attn_scale) {}
virtual void set_weight_adapter(const std::shared_ptr<WeightAdapter>& adapter) {}
virtual void runner_end() {}
};
@@ -157,7 +161,7 @@ public:
// Ref: https://github.com/AUTOMATIC1111/stable-diffusion-webui/blob/cad87bf4e3e0b0a759afa94e933527c3123d59bc/modules/sd_hijack_clip.py#L283
struct FrozenCLIPEmbedderWithCustomWords : public Conditioner {
SDVersion version = VERSION_SD1;
CLIPTokenizer tokenizer;
std::shared_ptr<Tokenizer> tokenizer;
std::shared_ptr<CLIPTextModelRunner> text_model;
std::shared_ptr<CLIPTextModelRunner> text_model2;
@@ -171,12 +175,18 @@ struct FrozenCLIPEmbedderWithCustomWords : public Conditioner {
const String2TensorStorage& tensor_storage_map,
const std::map<std::string, std::string>& orig_embedding_map,
SDVersion version = VERSION_SD1,
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr)
: version(version), tokenizer(sd_version_is_sd2(version) ? 0 : 49407) {
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr,
const TokenizerConfig& tokenizers = {})
: version(version) {
const int pad_id = sd_version_is_sd2(version) ? 0 : 49407;
tokenizer = tokenizers.create(TokenizerConfig::MAIN, 49408, pad_id, false, true);
if (!tokenizer) {
tokenizer = std::make_shared<CLIPTokenizer>(pad_id);
}
for (const auto& kv : orig_embedding_map) {
std::string name = normalize_embedding_name(kv.first);
embedding_map[name] = kv.second;
tokenizer.add_special_token(name);
tokenizer->add_special_token(name);
}
bool force_clip_f32 = !embedding_map.empty();
if (sd_version_is_sd1(version)) {
@@ -231,6 +241,13 @@ struct FrozenCLIPEmbedderWithCustomWords : public Conditioner {
}
}
void set_scale_overrides(float linear_scale, float attn_scale) override {
text_model->set_scale_overrides(linear_scale, attn_scale);
if (sd_version_is_sdxl(version)) {
text_model2->set_scale_overrides(linear_scale, attn_scale);
}
}
void set_weight_adapter(const std::shared_ptr<WeightAdapter>& adapter) override {
text_model->set_weight_adapter(adapter);
if (sd_version_is_sdxl(version)) {
@@ -356,16 +373,15 @@ struct FrozenCLIPEmbedderWithCustomWords : public Conditioner {
return load_embedding(name, iter->second, bpe_tokens);
}
std::vector<int> convert_token_to_id(std::string text) {
bool convert_token_to_id(const std::string& text, std::vector<int>& tokens) {
auto on_new_token_cb = [&](std::string& str, std::vector<int32_t>& bpe_tokens) -> bool {
return append_embedding_tokens(str, bpe_tokens);
};
std::vector<int> curr_tokens = tokenizer.encode(text, on_new_token_cb);
return curr_tokens;
return tokenizer->encode(text, tokens, on_new_token_cb);
}
std::string decode(const std::vector<int>& tokens) {
return tokenizer.decode(tokens);
bool decode(const std::vector<int>& tokens, std::string& text) {
return tokenizer->decode(tokens, text);
}
std::pair<std::vector<int>, std::vector<float>> tokenize(std::string text,
@@ -403,18 +419,21 @@ struct FrozenCLIPEmbedderWithCustomWords : public Conditioner {
if (padding_size > 0) {
LOG_VERBOSE("BREAK token encountered, padding current chunk by %zu tokens.", padding_size);
tokens.insert(tokens.end(), padding_size, tokenizer.EOS_TOKEN_ID);
tokens.insert(tokens.end(), padding_size, tokenizer->EOS_TOKEN_ID);
weights.insert(weights.end(), padding_size, 1.0f);
}
continue; // Skip to the next item after handling BREAK
}
std::vector<int> curr_tokens = tokenizer.encode(curr_text, on_new_token_cb);
std::vector<int> curr_tokens;
if (!tokenizer->encode(curr_text, curr_tokens, on_new_token_cb)) {
return {};
}
tokens.insert(tokens.end(), curr_tokens.begin(), curr_tokens.end());
weights.insert(weights.end(), curr_tokens.size(), curr_weight);
}
tokenizer.pad_tokens(tokens, &weights, nullptr, min_length, max_length, allow_overflow_expand);
tokenizer->pad_tokens(tokens, &weights, nullptr, min_length, max_length, allow_overflow_expand);
// for (int i = 0; i < tokens.size(); i++) {
// std::cout << tokens[i] << ":" << weights[i] << ", ";
@@ -451,7 +470,7 @@ struct FrozenCLIPEmbedderWithCustomWords : public Conditioner {
sd::Tensor<int32_t> input_ids2;
size_t max_token_idx = 0;
if (sd_version_is_sdxl(version)) {
auto it = std::find(chunk_tokens.begin(), chunk_tokens.end(), tokenizer.EOS_TOKEN_ID);
auto it = std::find(chunk_tokens.begin(), chunk_tokens.end(), tokenizer->EOS_TOKEN_ID);
if (it != chunk_tokens.end()) {
std::fill(std::next(it), chunk_tokens.end(), 0);
}
@@ -552,7 +571,10 @@ struct FrozenCLIPEmbedderWithCustomWords : public Conditioner {
SDCondition get_learned_condition(int n_threads,
const ConditionerParams& conditioner_params) override {
auto tokens_and_weights = tokenize(conditioner_params.text, text_model->model.n_token, text_model->model.n_token, true);
auto tokens_and_weights = tokenize(conditioner_params.text, text_model->model.n_token, text_model->model.n_token, true);
if (tokens_and_weights.first.empty()) {
return {};
}
std::vector<int>& tokens = tokens_and_weights.first;
std::vector<float>& weights = tokens_and_weights.second;
return get_learned_condition_common(n_threads,
@@ -620,8 +642,8 @@ struct FrozenCLIPVisionEmbedder : public GGMLRunner {
};
struct SD3CLIPEmbedder : public Conditioner {
CLIPTokenizer clip_l_tokenizer;
CLIPTokenizer clip_g_tokenizer;
std::shared_ptr<Tokenizer> clip_l_tokenizer;
std::shared_ptr<Tokenizer> clip_g_tokenizer;
T5UniGramTokenizer t5_tokenizer;
std::shared_ptr<CLIPTextModelRunner> clip_l;
std::shared_ptr<CLIPTextModelRunner> clip_g;
@@ -629,8 +651,8 @@ struct SD3CLIPEmbedder : public Conditioner {
SD3CLIPEmbedder(ggml_backend_t backend,
const String2TensorStorage& tensor_storage_map = {},
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr)
: clip_g_tokenizer(0) {
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr,
const TokenizerConfig& tokenizers = {}) {
bool use_clip_l = false;
bool use_clip_g = false;
bool use_t5 = false;
@@ -648,9 +670,17 @@ struct SD3CLIPEmbedder : public Conditioner {
return;
}
if (use_clip_l) {
clip_l_tokenizer = tokenizers.create(TokenizerConfig::CLIP_L, 49408, 49407, false, true);
if (!clip_l_tokenizer) {
clip_l_tokenizer = std::make_shared<CLIPTokenizer>();
}
clip_l = std::make_shared<CLIPTextModelRunner>(backend, tensor_storage_map, "text_encoders.clip_l.transformer.text_model", OPENAI_CLIP_VIT_L_14, false, false, weight_manager);
}
if (use_clip_g) {
clip_g_tokenizer = tokenizers.create(TokenizerConfig::CLIP_G, 49408, 0, false, true);
if (!clip_g_tokenizer) {
clip_g_tokenizer = std::make_shared<CLIPTokenizer>(0);
}
clip_g = std::make_shared<CLIPTextModelRunner>(backend, tensor_storage_map, "text_encoders.clip_g.transformer.text_model", OPEN_CLIP_VIT_BIGG_14, false, false, weight_manager);
}
if (use_t5) {
@@ -736,6 +766,18 @@ struct SD3CLIPEmbedder : public Conditioner {
}
}
void set_scale_overrides(float linear_scale, float attn_scale) override {
if (clip_l) {
clip_l->set_scale_overrides(linear_scale, attn_scale);
}
if (clip_g) {
clip_g->set_scale_overrides(linear_scale, attn_scale);
}
if (t5) {
t5->set_scale_overrides(linear_scale, attn_scale);
}
}
void set_weight_adapter(const std::shared_ptr<WeightAdapter>& adapter) override {
if (clip_l) {
clip_l->set_weight_adapter(adapter);
@@ -790,27 +832,36 @@ struct SD3CLIPEmbedder : public Conditioner {
const std::string& curr_text = item.first;
float curr_weight = item.second;
if (clip_l) {
std::vector<int> curr_tokens = clip_l_tokenizer.encode(curr_text, on_new_token_cb);
std::vector<int> curr_tokens;
if (!clip_l_tokenizer->encode(curr_text, curr_tokens, on_new_token_cb)) {
return {};
}
clip_l_tokens.insert(clip_l_tokens.end(), curr_tokens.begin(), curr_tokens.end());
clip_l_weights.insert(clip_l_weights.end(), curr_tokens.size(), curr_weight);
}
if (clip_g) {
std::vector<int> curr_tokens = clip_g_tokenizer.encode(curr_text, on_new_token_cb);
std::vector<int> curr_tokens;
if (!clip_g_tokenizer->encode(curr_text, curr_tokens, on_new_token_cb)) {
return {};
}
clip_g_tokens.insert(clip_g_tokens.end(), curr_tokens.begin(), curr_tokens.end());
clip_g_weights.insert(clip_g_weights.end(), curr_tokens.size(), curr_weight);
}
if (t5) {
std::vector<int> curr_tokens = t5_tokenizer.encode(curr_text);
std::vector<int> curr_tokens;
if (!t5_tokenizer.encode(curr_text, curr_tokens)) {
return {};
}
t5_tokens.insert(t5_tokens.end(), curr_tokens.begin(), curr_tokens.end());
t5_weights.insert(t5_weights.end(), curr_tokens.size(), curr_weight);
}
}
if (clip_l) {
clip_l_tokenizer.pad_tokens(clip_l_tokens, &clip_l_weights, nullptr, min_length, max_length, allow_overflow_expand);
clip_l_tokenizer->pad_tokens(clip_l_tokens, &clip_l_weights, nullptr, min_length, max_length, allow_overflow_expand);
}
if (clip_g) {
clip_g_tokenizer.pad_tokens(clip_g_tokens, &clip_g_weights, nullptr, min_length, max_length, allow_overflow_expand);
clip_g_tokenizer->pad_tokens(clip_g_tokens, &clip_g_weights, nullptr, min_length, max_length, allow_overflow_expand);
}
if (t5) {
t5_tokenizer.pad_tokens(t5_tokens, &t5_weights, nullptr, min_length, max_length, true);
@@ -881,7 +932,7 @@ struct SD3CLIPEmbedder : public Conditioner {
chunk_hidden_states_l = ::apply_token_weights(std::move(chunk_hidden_states_l), chunk_weights);
if (chunk_idx == 0) {
auto it = std::find(chunk_tokens.begin(), chunk_tokens.end(), clip_l_tokenizer.EOS_TOKEN_ID);
auto it = std::find(chunk_tokens.begin(), chunk_tokens.end(), clip_l_tokenizer->EOS_TOKEN_ID);
max_token_idx = std::min<size_t>(std::distance(chunk_tokens.begin(), it), chunk_tokens.size() - 1);
pooled_l = clip_l->compute(n_threads,
input_ids,
@@ -924,7 +975,7 @@ struct SD3CLIPEmbedder : public Conditioner {
chunk_hidden_states_g = ::apply_token_weights(std::move(chunk_hidden_states_g), chunk_weights);
if (chunk_idx == 0) {
auto it = std::find(chunk_tokens.begin(), chunk_tokens.end(), clip_g_tokenizer.EOS_TOKEN_ID);
auto it = std::find(chunk_tokens.begin(), chunk_tokens.end(), clip_g_tokenizer->EOS_TOKEN_ID);
max_token_idx = std::min<size_t>(std::distance(chunk_tokens.begin(), it), chunk_tokens.size() - 1);
pooled_g = clip_g->compute(n_threads,
input_ids,
@@ -1002,6 +1053,9 @@ struct SD3CLIPEmbedder : public Conditioner {
SDCondition get_learned_condition(int n_threads,
const ConditionerParams& conditioner_params) override {
auto tokens_and_weights = tokenize(conditioner_params.text, 77, 77, true);
if (tokens_and_weights.empty()) {
return {};
}
return get_learned_condition_common(n_threads,
tokens_and_weights,
conditioner_params.clip_skip,
@@ -1010,7 +1064,7 @@ struct SD3CLIPEmbedder : public Conditioner {
};
struct FluxCLIPEmbedder : public Conditioner {
CLIPTokenizer clip_l_tokenizer;
std::shared_ptr<Tokenizer> clip_l_tokenizer;
T5UniGramTokenizer t5_tokenizer;
std::shared_ptr<CLIPTextModelRunner> clip_l;
std::shared_ptr<T5Runner> t5;
@@ -1018,7 +1072,8 @@ struct FluxCLIPEmbedder : public Conditioner {
FluxCLIPEmbedder(ggml_backend_t backend,
const String2TensorStorage& tensor_storage_map = {},
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr) {
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr,
const TokenizerConfig& tokenizers = {}) {
bool use_clip_l = false;
bool use_t5 = false;
for (auto pair : tensor_storage_map) {
@@ -1035,6 +1090,11 @@ struct FluxCLIPEmbedder : public Conditioner {
}
if (use_clip_l) {
auto slot = tokenizers.has(TokenizerConfig::CLIP_L) ? TokenizerConfig::CLIP_L : TokenizerConfig::MAIN;
clip_l_tokenizer = tokenizers.create(slot, 49408, 49407, false, true);
if (!clip_l_tokenizer) {
clip_l_tokenizer = std::make_shared<CLIPTokenizer>();
}
clip_l = std::make_shared<CLIPTextModelRunner>(backend, tensor_storage_map, "text_encoders.clip_l.transformer.text_model", OPENAI_CLIP_VIT_L_14, true, false, weight_manager);
} else {
LOG_WARN("clip_l text encoder not found! Prompt adherence might be degraded.");
@@ -1106,6 +1166,15 @@ struct FluxCLIPEmbedder : public Conditioner {
}
}
void set_scale_overrides(float linear_scale, float attn_scale) override {
if (clip_l) {
clip_l->set_scale_overrides(linear_scale, attn_scale);
}
if (t5) {
t5->set_scale_overrides(linear_scale, attn_scale);
}
}
void set_weight_adapter(const std::shared_ptr<WeightAdapter>& adapter) override {
if (clip_l) {
clip_l->set_weight_adapter(adapter);
@@ -1151,19 +1220,25 @@ struct FluxCLIPEmbedder : public Conditioner {
const std::string& curr_text = item.first;
float curr_weight = item.second;
if (clip_l) {
std::vector<int> curr_tokens = clip_l_tokenizer.encode(curr_text, on_new_token_cb);
std::vector<int> curr_tokens;
if (!clip_l_tokenizer->encode(curr_text, curr_tokens, on_new_token_cb)) {
return {};
}
clip_l_tokens.insert(clip_l_tokens.end(), curr_tokens.begin(), curr_tokens.end());
clip_l_weights.insert(clip_l_weights.end(), curr_tokens.size(), curr_weight);
}
if (t5) {
std::vector<int> curr_tokens = t5_tokenizer.encode(curr_text);
std::vector<int> curr_tokens;
if (!t5_tokenizer.encode(curr_text, curr_tokens)) {
return {};
}
t5_tokens.insert(t5_tokens.end(), curr_tokens.begin(), curr_tokens.end());
t5_weights.insert(t5_weights.end(), curr_tokens.size(), curr_weight);
}
}
if (clip_l) {
clip_l_tokenizer.pad_tokens(clip_l_tokens, &clip_l_weights, nullptr, 77, 77, true);
clip_l_tokenizer->pad_tokens(clip_l_tokens, &clip_l_weights, nullptr, 77, 77, true);
}
if (t5) {
t5_tokenizer.pad_tokens(t5_tokens, &t5_weights, nullptr, min_length, max_length, true);
@@ -1213,7 +1288,7 @@ struct FluxCLIPEmbedder : public Conditioner {
sd::Tensor<int32_t> input_ids({static_cast<int64_t>(chunk_tokens.size())}, chunk_tokens);
size_t max_token_idx = 0;
auto it = std::find(chunk_tokens.begin(), chunk_tokens.end(), clip_l_tokenizer.EOS_TOKEN_ID);
auto it = std::find(chunk_tokens.begin(), chunk_tokens.end(), clip_l_tokenizer->EOS_TOKEN_ID);
max_token_idx = std::min<size_t>(std::distance(chunk_tokens.begin(), it), chunk_tokens.size() - 1);
pooled = clip_l->compute(n_threads,
@@ -1224,7 +1299,10 @@ struct FluxCLIPEmbedder : public Conditioner {
true,
clip_skip,
false);
GGML_ASSERT(!pooled.empty());
if (pooled.empty()) {
LOG_ERROR("Flux CLIP-L encoding failed");
return {};
}
} else {
pooled = sd::Tensor<float>::zeros({768});
}
@@ -1243,7 +1321,10 @@ struct FluxCLIPEmbedder : public Conditioner {
input_ids,
sd::Tensor<float>(),
false);
GGML_ASSERT(!chunk_hidden_states.empty());
if (chunk_hidden_states.empty()) {
LOG_ERROR("Flux T5 encoding failed at chunk %d/%zu", chunk_idx + 1, chunk_count);
return {};
}
chunk_hidden_states = ::apply_token_weights(std::move(chunk_hidden_states), chunk_weights);
if (zero_out_masked) {
chunk_hidden_states.fill_(0.0f);
@@ -1270,6 +1351,9 @@ struct FluxCLIPEmbedder : public Conditioner {
SDCondition get_learned_condition(int n_threads,
const ConditionerParams& conditioner_params) override {
auto tokens_and_weights = tokenize(conditioner_params.text, chunk_len, chunk_len);
if (tokens_and_weights.empty()) {
return {};
}
return get_learned_condition_common(n_threads,
tokens_and_weights,
conditioner_params.clip_skip,
@@ -1368,6 +1452,12 @@ struct T5CLIPEmbedder : public Conditioner {
}
}
void set_scale_overrides(float linear_scale, float attn_scale) override {
if (t5) {
t5->set_scale_overrides(linear_scale, attn_scale);
}
}
void set_weight_adapter(const std::shared_ptr<WeightAdapter>& adapter) override {
if (t5) {
t5->set_weight_adapter(adapter);
@@ -1407,7 +1497,10 @@ struct T5CLIPEmbedder : public Conditioner {
const std::string& curr_text = item.first;
float curr_weight = item.second;
std::vector<int> curr_tokens = t5_tokenizer.encode(curr_text);
std::vector<int> curr_tokens;
if (!t5_tokenizer.encode(curr_text, curr_tokens)) {
return {};
}
t5_tokens.insert(t5_tokens.end(), curr_tokens.begin(), curr_tokens.end());
t5_weights.insert(t5_weights.end(), curr_tokens.size(), curr_weight);
}
@@ -1505,6 +1598,9 @@ struct T5CLIPEmbedder : public Conditioner {
SDCondition get_learned_condition(int n_threads,
const ConditionerParams& conditioner_params) override {
auto tokens_and_weights = tokenize(conditioner_params.text, chunk_len, chunk_len);
if (std::get<0>(tokens_and_weights).empty()) {
return {};
}
return get_learned_condition_common(n_threads,
tokens_and_weights,
conditioner_params.clip_skip,
@@ -1576,6 +1672,12 @@ struct MiniT2IConditioner : public Conditioner {
}
}
void set_scale_overrides(float linear_scale, float attn_scale) override {
if (t5) {
t5->set_scale_overrides(linear_scale, attn_scale);
}
}
void set_weight_adapter(const std::shared_ptr<WeightAdapter>& adapter) override {
if (t5) {
t5->set_weight_adapter(adapter);
@@ -1597,7 +1699,10 @@ struct MiniT2IConditioner : public Conditioner {
return result;
}
std::vector<int> tokens = tokenizer.encode(conditioner_params.text);
std::vector<int> tokens;
if (!tokenizer.encode(conditioner_params.text, tokens)) {
return {};
}
if (tokens.size() > prompt_length) {
tokens.resize(prompt_length);
}
@@ -1623,21 +1728,93 @@ struct MiniT2IConditioner : public Conditioner {
}
};
struct SenseNovaU1Conditioner : public Conditioner {
static constexpr size_t kMaxPromptTokens = 12288;
SenseNovaU1Tokenizer tokenizer;
void get_param_tensors(std::map<std::string, ggml_tensor*>& tensors) override {
SD_UNUSED(tensors);
}
void set_flash_attention_enabled(bool enabled) override {
SD_UNUSED(enabled);
}
static std::string build_query(const std::string& text, bool is_negative) {
static const std::string kSystemMessage =
"You are an image generation and editing assistant that accurately understands and executes user intent.\n\n"
"You support two modes:\n\n1. Think Mode:\nIf the task requires reasoning, you MUST start with a "
"<think></think> block. Put all reasoning inside the block using plain text. DO NOT include any image tags. "
"Keep it reasonable and directly useful for producing the final image.\n\n2. Non-Think Mode:\nIf no reasoning "
"is needed, directly produce the final image.\n\nTask Types:\n\nA. Text-to-Image Generation:\n- Generate a "
"high-quality image based on the user's description.\n- Ensure visual clarity, semantic consistency, and "
"completeness.\n- DO NOT introduce elements that contradict or override the user's intent.\n\nB. Image Editing:\n"
"- Use the provided image(s) as input or reference for modification or transformation.\n- The result can be an "
"edited image or a new image based on the reference(s).\n- Preserve all unspecified attributes unless explicitly "
"changed.\n\nGeneral Rules:\n- For any visible text in the image, follow the language specified for the rendered "
"text in the user's description, not the language of the prompt. If no language is specified, use the user's input "
"language.";
std::string query;
if (!is_negative) {
query += "<|im_start|>system\n";
query += kSystemMessage;
query += "<|im_end|>\n";
}
query += "<|im_start|>user\n";
query += text;
query += "<|im_end|>\n<|im_start|>assistant\n";
query += is_negative ? "<img>" : "<think>\n\n</think>\n\n<img>";
return query;
}
SDCondition tokenize_condition(const std::string& text, bool is_negative) {
std::vector<int> tokens;
if (!tokenizer.encode(build_query(text, is_negative), tokens)) {
return {};
}
if (tokens.empty() || tokens.size() > kMaxPromptTokens) {
LOG_ERROR("SenseNova U1.5 prompt token count %zu is outside [1, %zu]",
tokens.size(),
kMaxPromptTokens);
return {};
}
SDCondition result;
result.c_input_ids = sd::Tensor<int32_t>({static_cast<int64_t>(tokens.size())}, tokens);
return result;
}
SDCondition get_learned_condition(int n_threads,
const ConditionerParams& conditioner_params) override {
SD_UNUSED(n_threads);
return tokenize_condition(conditioner_params.text, false);
}
SDCondition get_unconditional_condition(const std::string& text) {
return tokenize_condition(text, true);
}
};
struct AnimaConditioner : public Conditioner {
std::shared_ptr<BPETokenizer> qwen_tokenizer;
std::shared_ptr<Tokenizer> qwen_tokenizer;
T5UniGramTokenizer t5_tokenizer;
std::shared_ptr<LLM::LLMRunner> llm;
AnimaConditioner(ggml_backend_t backend,
const String2TensorStorage& tensor_storage_map = {},
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr) {
qwen_tokenizer = std::make_shared<Qwen2Tokenizer>();
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr,
const TokenizerConfig& tokenizers = {}) {
llm = std::make_shared<LLM::LLMRunner>(LLM::LLMArch::QWEN3,
backend,
tensor_storage_map,
"text_encoders.llm",
false,
weight_manager);
qwen_tokenizer = tokenizers.create(TokenizerConfig::MAIN, llm->config.vocab_size, 151643);
if (!qwen_tokenizer) {
qwen_tokenizer = std::make_shared<Qwen2Tokenizer>();
}
}
void get_param_tensors(std::map<std::string, ggml_tensor*>& tensors) override {
@@ -1672,6 +1849,10 @@ struct AnimaConditioner : public Conditioner {
llm->set_flash_attention_enabled(enabled);
}
void set_scale_overrides(float linear_scale, float attn_scale) override {
llm->set_scale_overrides(linear_scale, attn_scale);
}
void set_weight_adapter(const std::shared_ptr<WeightAdapter>& adapter) override {
llm->set_weight_adapter(adapter);
}
@@ -1700,7 +1881,10 @@ struct AnimaConditioner : public Conditioner {
for (const auto& item : parsed_attention) {
const std::string& curr_text = item.first;
std::vector<int> curr_tokens = qwen_tokenizer->tokenize(curr_text, nullptr);
std::vector<int> curr_tokens;
if (!qwen_tokenizer->tokenize(curr_text, curr_tokens, nullptr)) {
return {};
}
qwen_tokens.insert(qwen_tokens.end(), curr_tokens.begin(), curr_tokens.end());
// Anima uses uniform Qwen token weights.
qwen_weights.insert(qwen_weights.end(), curr_tokens.size(), 1.f);
@@ -1713,7 +1897,10 @@ struct AnimaConditioner : public Conditioner {
for (const auto& item : parsed_attention) {
const std::string& curr_text = item.first;
float curr_weight = item.second;
std::vector<int> curr_tokens = t5_tokenizer.encode(curr_text);
std::vector<int> curr_tokens;
if (!t5_tokenizer.encode(curr_text, curr_tokens)) {
return {};
}
t5_tokens.insert(t5_tokens.end(), curr_tokens.begin(), curr_tokens.end());
t5_weights.insert(t5_weights.end(), curr_tokens.size(), curr_weight);
}
@@ -1732,6 +1919,10 @@ struct AnimaConditioner : public Conditioner {
auto& t5_tokens = std::get<2>(tokenized);
auto& t5_weights = std::get<3>(tokenized);
if (qwen_tokens.empty()) {
return {};
}
sd::Tensor<int32_t> input_ids({static_cast<int64_t>(qwen_tokens.size()), 1}, qwen_tokens);
auto hidden_states = llm->compute(n_threads,
input_ids,
@@ -1758,7 +1949,7 @@ struct AnimaConditioner : public Conditioner {
struct LLMEmbedder : public Conditioner {
SDVersion version;
std::shared_ptr<BPETokenizer> tokenizer;
std::shared_ptr<Tokenizer> tokenizer;
std::shared_ptr<LLM::LLMRunner> llm;
std::shared_ptr<T5Runner> byt5;
@@ -1767,8 +1958,17 @@ struct LLMEmbedder : public Conditioner {
SDVersion version = VERSION_QWEN_IMAGE,
const std::string prefix = "",
bool enable_vision = false,
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr)
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr,
const TokenizerConfig& tokenizers = {})
: version(version) {
if (!tokenizers.has(TokenizerConfig::MAIN)) {
if (sd_version_is_lens(version)) {
throw std::runtime_error("Lens requires an external GPT-OSS tokenizer.json; pass --tokenizer FILE or set sd_ctx_params_t::tokenizer");
}
if (sd_version_is_pid(version)) {
throw std::runtime_error("PiD requires an external Gemma 2 tokenizer.json; pass --tokenizer FILE or set sd_ctx_params_t::tokenizer");
}
}
LLM::LLMArch arch = LLM::LLMArch::QWEN2_5_VL;
if (version == VERSION_FLUX2) {
arch = LLM::LLMArch::MISTRAL_SMALL_3_2;
@@ -1778,7 +1978,8 @@ struct LLMEmbedder : public Conditioner {
arch = LLM::LLMArch::GPT_OSS_20B;
} else if (sd_version_is_pid(version)) {
arch = LLM::LLMArch::GEMMA2_2B;
} else if (sd_version_is_lingbot_video(version) ||
} else if (version == VERSION_QWEN_IMAGE_2_1 ||
sd_version_is_lingbot_video(version) ||
sd_version_is_ideogram4(version) ||
sd_version_is_boogu_image(version) ||
sd_version_is_sefi_image(version) ||
@@ -1789,21 +1990,28 @@ struct LLMEmbedder : public Conditioner {
} else if (sd_version_is_z_image(version) || version == VERSION_OVIS_IMAGE || version == VERSION_FLUX2_KLEIN) {
arch = LLM::LLMArch::QWEN3;
}
if (arch == LLM::LLMArch::MISTRAL_SMALL_3_2 || arch == LLM::LLMArch::MINISTRAL_3_3B) {
tokenizer = std::make_shared<MistralTokenizer>();
} else if (arch == LLM::LLMArch::GPT_OSS_20B) {
tokenizer = std::make_shared<GPTOSSTokenizer>();
} else if (arch == LLM::LLMArch::GEMMA2_2B) {
tokenizer = std::make_shared<Gemma2Tokenizer>();
} else {
tokenizer = std::make_shared<Qwen2Tokenizer>();
}
llm = std::make_shared<LLM::LLMRunner>(arch,
llm = std::make_shared<LLM::LLMRunner>(arch,
backend,
tensor_storage_map,
"text_encoders.llm",
enable_vision,
weight_manager);
int pad_id = 151643;
if (arch == LLM::LLMArch::MISTRAL_SMALL_3_2 || arch == LLM::LLMArch::MINISTRAL_3_3B) {
pad_id = 11;
} else if (arch == LLM::LLMArch::GPT_OSS_20B) {
pad_id = 199999;
} else if (arch == LLM::LLMArch::GEMMA2_2B) {
pad_id = 0;
}
tokenizer = tokenizers.create(TokenizerConfig::MAIN, llm->config.vocab_size, pad_id);
if (!tokenizer) {
if (arch == LLM::LLMArch::MISTRAL_SMALL_3_2 || arch == LLM::LLMArch::MINISTRAL_3_3B) {
tokenizer = std::make_shared<MistralTokenizer>();
} else {
tokenizer = std::make_shared<Qwen2Tokenizer>();
}
}
if (sd_version_is_hunyuan_video(version)) {
const std::string byt5_prefix = "text_encoders.t5xxl.transformer";
for (const auto& [name, _] : tensor_storage_map) {
@@ -1876,6 +2084,13 @@ struct LLMEmbedder : public Conditioner {
}
}
void set_scale_overrides(float linear_scale, float attn_scale) override {
llm->set_scale_overrides(linear_scale, attn_scale);
if (byt5) {
byt5->set_scale_overrides(linear_scale, attn_scale);
}
}
void set_weight_adapter(const std::shared_ptr<WeightAdapter>& adapter) override {
if (llm) {
llm->set_weight_adapter(adapter);
@@ -1935,7 +2150,10 @@ struct LLMEmbedder : public Conditioner {
for (const auto& item : parsed_attention) {
const std::string& curr_text = item.first;
float curr_weight = item.second;
std::vector<int> curr_tokens = tokenizer->encode(curr_text, nullptr);
std::vector<int> curr_tokens;
if (!tokenizer->encode(curr_text, curr_tokens, nullptr)) {
return {};
}
tokens.insert(tokens.end(), curr_tokens.begin(), curr_tokens.end());
weights.insert(weights.end(), curr_tokens.size(), curr_weight);
}
@@ -1968,6 +2186,10 @@ struct LLMEmbedder : public Conditioner {
auto& weights = std::get<1>(tokens_weights_mask);
auto& mask = std::get<2>(tokens_weights_mask);
if (tokens.empty()) {
return {};
}
sd::Tensor<int32_t> input_ids({static_cast<int64_t>(tokens.size())}, tokens);
sd::Tensor<float> attention_mask;
if (!mask.empty()) {
@@ -2127,7 +2349,11 @@ struct LLMEmbedder : public Conditioner {
GGML_ASSERT(image_outputs.size() == 4);
auto image_embed = std::move(image_outputs[0]);
prompt += "<|vision_start|>";
int image_embed_idx = static_cast<int>(tokenizer->encode(prompt, nullptr).size());
std::vector<int> prefix_tokens;
if (!tokenizer->encode(prompt, prefix_tokens, nullptr)) {
return false;
}
int image_embed_idx = static_cast<int>(prefix_tokens.size());
image_embeds.emplace_back(image_embed_idx, image_embed);
if (deepstack_image_embeds.empty()) {
deepstack_image_embeds.resize(image_outputs.size() - 1);
@@ -2143,6 +2369,7 @@ struct LLMEmbedder : public Conditioner {
prompt += placeholder;
}
prompt += "<|vision_end|>";
return true;
};
const auto* references = conditioner_params.minimax_h3_references;
@@ -2159,11 +2386,13 @@ struct LLMEmbedder : public Conditioner {
GGML_ASSERT(item.frames.size() == 1);
auto resized = resize_for_vision(item.frames[0]);
prompt += "<Picture " + std::to_string(++picture_index) + ">: ";
add_vision_outputs(llm->encode_image_outputs(n_threads,
resized,
false),
static_cast<int>(resized.shape()[1]) / patch_size,
static_cast<int>(resized.shape()[0]) / patch_size);
if (!add_vision_outputs(llm->encode_image_outputs(n_threads,
resized,
false),
static_cast<int>(resized.shape()[1]) / patch_size,
static_cast<int>(resized.shape()[0]) / patch_size)) {
return {};
}
continue;
}
@@ -2187,22 +2416,26 @@ struct LLMEmbedder : public Conditioner {
second.shape()[3]});
}
auto pair = sd::ops::concat(first.unsqueeze(2), second.unsqueeze(2), 2);
add_vision_outputs(llm->encode_video_block_outputs(n_threads,
pair,
false),
static_cast<int>(first.shape()[1]) / patch_size,
static_cast<int>(first.shape()[0]) / patch_size);
if (!add_vision_outputs(llm->encode_video_block_outputs(n_threads,
pair,
false),
static_cast<int>(first.shape()[1]) / patch_size,
static_cast<int>(first.shape()[0]) / patch_size)) {
return {};
}
}
}
} else if (conditioner_params.ref_images != nullptr) {
for (size_t i = 0; i < conditioner_params.ref_images->size(); ++i) {
auto resized = resize_for_vision((*conditioner_params.ref_images)[i]);
prompt += "<Picture " + std::to_string(i + 1) + ">: ";
add_vision_outputs(llm->encode_image_outputs(n_threads,
resized,
false),
static_cast<int>(resized.shape()[1]) / patch_size,
static_cast<int>(resized.shape()[0]) / patch_size);
if (!add_vision_outputs(llm->encode_image_outputs(n_threads,
resized,
false),
static_cast<int>(resized.shape()[1]) / patch_size,
static_cast<int>(resized.shape()[0]) / patch_size)) {
return {};
}
}
}
}
@@ -2238,7 +2471,10 @@ struct LLMEmbedder : public Conditioner {
"enhanced description for the prompt below and avoid including any additional "
"commentary or evaluations:<|im_end|>\n<|im_start|>user\n";
auto prefix_tokens = tokenizer->encode(prompt_prefix, nullptr);
std::vector<int> prefix_tokens;
if (!tokenizer->encode(prompt_prefix, prefix_tokens, nullptr)) {
return {};
}
prompt_template_encode_start_idx = 0;
for (int token : prefix_tokens) {
if (token != pad_token) {
@@ -2291,7 +2527,11 @@ struct LLMEmbedder : public Conditioner {
GGML_ASSERT(!image_embed.empty());
std::string image_prefix = prompt + img_prompt + "<|vision_start|>";
int image_embed_idx = static_cast<int>(tokenizer->encode(image_prefix, nullptr).size());
std::vector<int> prefix_tokens;
if (!tokenizer->encode(image_prefix, prefix_tokens, nullptr)) {
return {};
}
int image_embed_idx = static_cast<int>(prefix_tokens.size());
image_embeds.emplace_back(image_embed_idx, image_embed);
img_prompt += "<|vision_start|>";
@@ -2308,6 +2548,67 @@ struct LLMEmbedder : public Conditioner {
prompt += conditioner_params.text;
prompt_attn_range = {0, 0};
prompt += "<|im_end|>\n<|im_start|>assistant\n";
} else if (version == VERSION_QWEN_IMAGE_2_1) {
if (!llm->enable_vision && conditioner_params.ref_images != nullptr && !conditioner_params.ref_images->empty()) {
LOG_ERROR("Qwen Image 2.1 editing requires Qwen3-VL vision weights; provide --llm_vision or a combined encoder");
return {};
}
prompt = "<|im_start|>system\nComprehend and analyze the provided prompt.<|im_end|>\n";
std::vector<int> system_tokens;
if (!tokenizer->encode(prompt, system_tokens, nullptr)) {
return {};
}
prompt_template_encode_start_idx = static_cast<int>(system_tokens.size());
out_layers = {static_cast<int>(llm->config.num_layers)};
prompt += "<|im_start|>user\n";
if (llm->enable_vision && conditioner_params.ref_images != nullptr) {
for (size_t i = 0; i < conditioner_params.ref_images->size(); ++i) {
const auto& image = (*conditioner_params.ref_images)[i];
int64_t width = image.shape()[0];
int64_t height = image.shape()[1];
int64_t pixels = width * height;
if (width % 32 != 0 || height % 32 != 0) {
LOG_ERROR("Qwen Image 2.1 reference dimensions must be multiples of 32");
return {};
}
auto rgb = sd::Tensor<float>({width, height, 3, 1});
for (int64_t p = 0; p < pixels; ++p) {
float alpha = image.shape()[2] == 4 ? image[p + 3 * pixels] : 1.f;
for (int c = 0; c < 3; ++c) {
rgb[p + c * pixels] = 2.f * (image[p + c * pixels] * alpha + 1.f - alpha) - 1.f;
}
}
auto outputs = llm->encode_image_outputs(n_threads, rgb, false);
if (outputs.empty()) {
return {};
}
prompt += (i == 0 ? "" : " ") + std::string("<image") + std::to_string(i + 1) + "><|vision_start|>";
std::vector<int> prefix_tokens;
if (!tokenizer->encode(prompt, prefix_tokens, nullptr)) {
return {};
}
int index = static_cast<int>(prefix_tokens.size());
int count = static_cast<int>(outputs[0].shape()[1]);
image_embeds.emplace_back(index, std::move(outputs[0]));
if (deepstack_image_embeds.empty()) {
deepstack_image_embeds.resize(outputs.size() - 1);
}
for (size_t layer = 1; layer < outputs.size(); ++layer) {
deepstack_image_embeds[layer - 1].emplace_back(index, std::move(outputs[layer]));
}
image_grids.push_back({index, count,
static_cast<int>(height) / llm->config.vision.patch_size,
static_cast<int>(width) / llm->config.vision.patch_size});
for (int j = 0; j < count; ++j) {
prompt += "<|image_pad|>";
}
prompt += "<|vision_end|>";
}
}
prompt_attn_range.first = static_cast<int>(prompt.size());
prompt += conditioner_params.text.empty() ? " " : conditioner_params.text;
prompt_attn_range.second = static_cast<int>(prompt.size());
prompt += "<|im_end|>\n<|im_start|>assistant\n";
} else if (sd_version_is_qwen_image(version) || sd_version_is_mage_flow(version)) {
if (llm->enable_vision && conditioner_params.ref_images != nullptr && !conditioner_params.ref_images->empty()) {
LOG_INFO("%s", sd_version_is_mage_flow(version) ? "MageFlowEditPipeline" : "QwenImageEditPlusPipeline");
@@ -2433,7 +2734,11 @@ struct LLMEmbedder : public Conditioner {
GGML_ASSERT(!image_embed.empty());
std::string image_prefix = prompt_prefix + img_prompt + "<|vision_start|>";
int image_embed_idx = static_cast<int>(tokenizer->encode(image_prefix, nullptr).size());
std::vector<int> prefix_tokens;
if (!tokenizer->encode(image_prefix, prefix_tokens, nullptr)) {
return {};
}
int image_embed_idx = static_cast<int>(prefix_tokens.size());
image_embeds.emplace_back(image_embed_idx, image_embed);
img_prompt += "<|vision_start|>";
@@ -2501,7 +2806,11 @@ struct LLMEmbedder : public Conditioner {
GGML_ASSERT(!image_embed.empty());
std::string image_prefix = prompt + img_prompt + "Picture " + std::to_string(i + 1) + ": <|vision_start|>";
int image_embed_idx = static_cast<int>(tokenizer->encode(image_prefix, nullptr).size());
std::vector<int> prefix_tokens;
if (!tokenizer->encode(image_prefix, prefix_tokens, nullptr)) {
return {};
}
int image_embed_idx = static_cast<int>(prefix_tokens.size());
image_embeds.emplace_back(image_embed_idx, image_embed);
img_prompt += "Picture " + std::to_string(i + 1) + ": <|vision_start|>";
@@ -2706,7 +3015,10 @@ struct LLMEmbedder : public Conditioner {
"- User Prompt: A busy city street -> Enhanced: A bustling city street scene at dusk, featuring glowing street lamps, a diverse crowd of people in colorful clothing, and a double-decker bus passing by towering glass skyscrapers.\n"
"Please generate only the enhanced description for the prompt below and avoid including any additional commentary or evaluations:\n"
"User Prompt: ";
auto chi_tokens = std::get<0>(tokenize(chi_prompt, {0, 0}));
auto chi_tokens = std::get<0>(tokenize(chi_prompt, {0, 0}));
if (chi_tokens.empty()) {
return {};
}
size_t num_chi_tokens = chi_tokens.size();
max_length = (int)num_chi_tokens + pixeldit_max_length - 2;
min_length = max_length;
@@ -2725,7 +3037,9 @@ struct LLMEmbedder : public Conditioner {
0,
false,
max_length);
GGML_ASSERT(!hidden_states.empty());
if (hidden_states.empty()) {
return {};
}
if (hidden_states.shape()[1] > pixeldit_max_length) {
auto bos = sd::ops::slice(hidden_states, 1, 0, 1);
@@ -2758,6 +3072,9 @@ struct LLMEmbedder : public Conditioner {
max_length,
deepstack_image_embeds,
image_grids);
if (hidden_states.empty()) {
return {};
}
std::vector<sd::Tensor<float>> extra_hidden_states_vec;
if (sd_version_is_hunyuan_video(version) && byt5) {
std::vector<std::string> quoted_texts;
@@ -2808,6 +3125,9 @@ struct LLMEmbedder : public Conditioner {
prompt_template_encode_start_idx,
spell_quotes,
max_length);
if (extra_hidden_states.empty()) {
return {};
}
extra_hidden_states_vec.push_back(std::move(extra_hidden_states));
}
@@ -2816,6 +3136,21 @@ struct LLMEmbedder : public Conditioner {
SDCondition result;
result.c_crossattn = std::move(hidden_states);
result.extra_c_crossattns = std::move(extra_hidden_states_vec);
if (version == VERSION_QWEN_IMAGE_2_1) {
auto slots = sd::Tensor<int32_t>::zeros({result.c_crossattn.shape()[1]});
for (size_t i = 0; i < image_embeds.size(); ++i) {
int64_t begin = image_embeds[i].first - prompt_template_encode_start_idx;
int64_t end = begin + image_embeds[i].second.shape()[1];
if (begin < 0 || end > slots.numel()) {
LOG_ERROR("Qwen Image 2.1 image slots exceed the encoded prompt");
return {};
}
for (int64_t j = begin; j < end; ++j) {
slots[j] = static_cast<int32_t>(i + 1);
}
}
result.c_token_types = std::move(slots);
}
if (sd_version_is_minimax_h3(version)) {
std::vector<int32_t> tags(static_cast<size_t>(result.c_crossattn.shape()[1]), 1);
for (const auto& [index, image_embed] : image_embeds) {
@@ -2906,7 +3241,7 @@ struct LTXAVEmbedder : public Conditioner {
static constexpr int64_t kNumStates = 49;
static constexpr int64_t kMinLength = 1024;
std::shared_ptr<GemmaTokenizer> tokenizer;
std::shared_ptr<Tokenizer> tokenizer;
std::shared_ptr<LLM::LLMRunner> llm;
std::shared_ptr<LTXAVTextProjectionRunner> projector;
std::string projector_prefix;
@@ -2933,17 +3268,21 @@ struct LTXAVEmbedder : public Conditioner {
const String2TensorStorage& tensor_storage_map = {},
const std::string& llm_prefix = "text_encoders.llm",
const std::string& projector_prefix = "text_embedding_projection",
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr)
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr,
const TokenizerConfig& tokenizers = {})
: projector_prefix(projector_prefix) {
LLM::LLMArch arch = detect_gemma_arch(tensor_storage_map, llm_prefix);
LOG_INFO("ltxav text encoder: %s", arch == LLM::LLMArch::GEMMA4_12B ? "gemma 4" : "gemma 3");
tokenizer = std::make_shared<GemmaTokenizer>();
llm = std::make_shared<LLM::LLMRunner>(arch,
llm = std::make_shared<LLM::LLMRunner>(arch,
backend,
tensor_storage_map,
llm_prefix,
false,
weight_manager);
tokenizer = tokenizers.create(TokenizerConfig::MAIN, llm->config.vocab_size, 0, true);
if (!tokenizer) {
tokenizer = std::make_shared<GemmaTokenizer>();
}
dual_projection = tensor_storage_map.find(projector_prefix + ".video_aggregate_embed.weight") != tensor_storage_map.end();
projector = std::make_shared<LTXAVTextProjectionRunner>(backend,
tensor_storage_map,
@@ -2965,6 +3304,11 @@ struct LTXAVEmbedder : public Conditioner {
projector->set_flash_attention_enabled(enabled);
}
void set_scale_overrides(float linear_scale, float attn_scale) override {
llm->set_scale_overrides(linear_scale, attn_scale);
projector->set_scale_overrides(linear_scale, attn_scale);
}
void set_max_graph_vram_bytes(size_t max_vram_bytes) override {
llm->set_max_graph_vram_bytes(max_vram_bytes);
projector->set_max_graph_vram_bytes(max_vram_bytes);
@@ -3017,7 +3361,10 @@ struct LTXAVEmbedder : public Conditioner {
std::vector<int> tokens;
std::vector<float> weights;
for (const auto& item : parsed_attention) {
auto curr_tokens = tokenizer->encode(item.first, nullptr);
std::vector<int> curr_tokens;
if (!tokenizer->encode(item.first, curr_tokens, nullptr)) {
return {};
}
tokens.insert(tokens.end(), curr_tokens.begin(), curr_tokens.end());
weights.insert(weights.end(), curr_tokens.size(), item.second);
}
@@ -3035,6 +3382,10 @@ struct LTXAVEmbedder : public Conditioner {
auto& weights = std::get<1>(tokens_weights_mask);
auto& mask = std::get<2>(tokens_weights_mask);
if (tokens.empty()) {
return {};
}
sd::Tensor<int32_t> input_ids({static_cast<int64_t>(tokens.size())}, std::vector<int32_t>(tokens.begin(), tokens.end()));
sd::Tensor<float> attention_mask;
if (!mask.empty()) {
@@ -3133,7 +3484,9 @@ struct LTXAVEmbedder : public Conditioner {
prompt_attn_range.second = static_cast<int>(prompt.size());
auto hidden_states = encode_prompt(n_threads, prompt, prompt_attn_range);
GGML_ASSERT(!hidden_states.empty());
if (hidden_states.empty()) {
return {};
}
int64_t t1 = ggml_time_ms();
LOG_VERBOSE("computing LTXAV condition graph completed, taking %" PRId64 " ms", t1 - t0);
+103
View File
@@ -0,0 +1,103 @@
#include "wan_audio.h"
#include <algorithm>
#include <cmath>
#include <cstddef>
namespace sd::wan_audio {
static BucketPlan plan_buckets(int audio_frames, int batch_frames, int video_rate, int fps) {
BucketPlan plan;
plan.audio_frames = audio_frames;
plan.batch_frames = batch_frames;
plan.video_rate = video_rate;
plan.fps = fps;
const double scale = static_cast<double>(video_rate) / fps;
// Keep a trailing chunk even when audio ends on a chunk boundary.
plan.num_chunks = static_cast<int>(audio_frames / (batch_frames * scale)) + 1;
plan.bucket_frames = plan.num_chunks * batch_frames;
plan.padded_audio_frames = static_cast<int>(
std::ceil(plan.bucket_frames / static_cast<double>(fps) * video_rate));
return plan;
}
// Match NumPy's round-half-even sampling.
static int bucket_source_frame(int bucket_frame, int video_rate, int fps) {
return static_cast<int>(std::nearbyint(static_cast<double>(bucket_frame) * video_rate / fps));
}
static int interpolated_frame_count(int in_frames, int input_fps, int output_fps) {
return static_cast<int>(in_frames / static_cast<double>(input_fps) * output_fps);
}
// Match PyTorch linear interpolation with align_corners=True.
static std::vector<float> linear_interpolate_frames(const std::vector<float>& in,
int num_layers,
int in_frames,
int dim,
int out_frames) {
std::vector<float> out(static_cast<size_t>(num_layers) * out_frames * dim, 0.0f);
if (in.empty() || in_frames <= 0 || out_frames <= 0 || num_layers <= 0 || dim <= 0) {
return out;
}
const double scale = out_frames > 1 ? static_cast<double>(in_frames - 1) / (out_frames - 1) : 0.0;
for (int layer = 0; layer < num_layers; ++layer) {
for (int out_i = 0; out_i < out_frames; ++out_i) {
const double pos = out_i * scale;
const int src0 = static_cast<int>(pos);
const int src1 = std::min(src0 + 1, in_frames - 1);
const float frac = static_cast<float>(pos - src0);
const float* in_row = &in[(static_cast<size_t>(layer) * in_frames + src0) * dim];
const float* in_next = &in[(static_cast<size_t>(layer) * in_frames + src1) * dim];
float* out_row = &out[(static_cast<size_t>(layer) * out_frames + out_i) * dim];
for (int d = 0; d < dim; ++d) {
out_row[d] = in_row[d] * (1.0f - frac) + in_next[d] * frac;
}
}
}
return out;
}
std::vector<float> build_audio_buckets(const float* stacked_states,
int num_layers,
int in_frames,
int dim,
int batch_frames,
BucketPlan* plan_out,
int input_fps,
int video_rate,
int fps) {
if (stacked_states == nullptr || num_layers <= 0 || in_frames <= 0 || dim <= 0 || batch_frames <= 0) {
return {};
}
const int audio_frames = interpolated_frame_count(in_frames, input_fps, video_rate);
if (audio_frames <= 0) {
return {};
}
const std::vector<float> interpolated =
linear_interpolate_frames(std::vector<float>(stacked_states,
stacked_states + static_cast<size_t>(num_layers) * in_frames * dim),
num_layers,
in_frames,
dim,
audio_frames);
const BucketPlan plan = plan_buckets(audio_frames, batch_frames, video_rate, fps);
if (plan_out != nullptr) {
*plan_out = plan;
}
std::vector<float> buckets(static_cast<size_t>(plan.bucket_frames) * num_layers * dim, 0.0f);
for (int frame = 0; frame < plan.bucket_frames; ++frame) {
const int src = bucket_source_frame(frame, video_rate, fps);
if (src >= plan.audio_frames) {
continue;
}
for (int layer = 0; layer < num_layers; ++layer) {
std::copy_n(interpolated.data() + (static_cast<size_t>(layer) * audio_frames + src) * dim,
static_cast<size_t>(dim),
buckets.data() + (static_cast<size_t>(frame) * num_layers + layer) * dim);
}
}
return buckets;
}
} // namespace sd::wan_audio
+32
View File
@@ -0,0 +1,32 @@
#ifndef __SD_CONDITIONING_WAN_AUDIO_H__
#define __SD_CONDITIONING_WAN_AUDIO_H__
#include <vector>
namespace sd::wan_audio {
struct BucketPlan {
int audio_frames; // frames at video_rate
int batch_frames; // latent_t * 4
int video_rate;
int fps; // bucket frame rate
int num_chunks; // includes trailing padding
int bucket_frames;
int padded_audio_frames;
};
// [layers, frames, dim] at input_fps -> [bucket_frames, layers, dim] at fps.
// Pads past the audio end; returns an empty vector on invalid input.
std::vector<float> build_audio_buckets(const float* stacked_states,
int num_layers,
int in_frames,
int dim,
int batch_frames,
BucketPlan* plan_out = nullptr,
int input_fps = 50,
int video_rate = 30,
int fps = 16);
} // namespace sd::wan_audio
#endif // __SD_CONDITIONING_WAN_AUDIO_H__
+3
View File
@@ -362,6 +362,9 @@ bool convert_with_components(const char* model_path,
const char* tensor_type_rules,
bool convert_name,
int n_threads) {
if (!validate_tensor_types(output_type, tensor_type_rules)) {
return false;
}
ModelLoader model_loader;
bool loaded_any = false;
+138 -41
View File
@@ -58,9 +58,15 @@ namespace sd::backend_fit {
size_t params_device = SIZE_MAX;
};
struct Runtime {
std::string name;
std::vector<size_t> devices;
};
struct Plan {
bool valid = false;
size_t main_device = SIZE_MAX;
std::vector<Runtime> runtimes;
std::vector<Decision> decisions;
};
@@ -96,7 +102,7 @@ namespace sd::backend_fit {
for (const auto& [name, stored_tensor] : loader.get_tensor_storage_map()) {
TensorStorage ts = stored_tensor;
ComponentKind kind;
if (is_unused_tensor(ts.name) || !classify_tensor(ts.name, kind)) {
if (!classify_tensor(ts.name, kind)) {
continue;
}
if (ts.expected_type != GGML_TYPE_COUNT) {
@@ -121,11 +127,14 @@ namespace sd::backend_fit {
return name;
}
static std::vector<Device> enumerate_gpu_devices(const sd::ggml_graph_cut::MaxVramAssignment& budgets) {
static std::vector<Device> enumerate_gpu_devices(const sd::ggml_graph_cut::MaxVramAssignment& budgets,
bool include_other_devices) {
std::vector<Device> out;
for (size_t i = 0; i < ggml_backend_dev_count(); ++i) {
ggml_backend_dev_t dev = ggml_backend_dev_get(i);
if (ggml_backend_dev_type(dev) != GGML_BACKEND_DEVICE_TYPE_GPU) {
const auto type = ggml_backend_dev_type(dev);
if (type != GGML_BACKEND_DEVICE_TYPE_GPU &&
(!include_other_devices || type == GGML_BACKEND_DEVICE_TYPE_CPU)) {
continue;
}
Device device;
@@ -183,18 +192,29 @@ namespace sd::backend_fit {
return -1;
}
static Plan compute_plan(const std::vector<Component>& components,
const std::vector<Device>& devices,
int64_t ram_budget_bytes) {
Plan plan;
static size_t select_main_device(const std::vector<Device>& devices) {
size_t main_device = SIZE_MAX;
for (size_t di = 0; di < devices.size(); ++di) {
if (devices[di].budget_bytes > 0 &&
(plan.main_device == SIZE_MAX || devices[di].budget_bytes > devices[plan.main_device].budget_bytes)) {
plan.main_device = di;
(main_device == SIZE_MAX || devices[di].budget_bytes > devices[main_device].budget_bytes)) {
main_device = di;
}
}
if (plan.main_device == SIZE_MAX) {
return plan;
return main_device;
}
static Plan compute_plan(const std::vector<Component>& components,
const std::vector<Device>& devices,
int64_t ram_budget_bytes,
const std::vector<Runtime>& runtimes = {}) {
Plan plan;
plan.main_device = select_main_device(devices);
plan.runtimes = runtimes;
if (plan.runtimes.empty()) {
if (plan.main_device == SIZE_MAX) {
return plan;
}
plan.runtimes.resize(components.size(), {devices[plan.main_device].name, {plan.main_device}});
}
std::vector<size_t> order(components.size());
@@ -212,6 +232,27 @@ namespace sd::backend_fit {
ram_budget_bytes = std::max<int64_t>(ram_budget_bytes, 0);
plan.decisions.resize(components.size());
auto uses_device = [&](size_t ci, size_t di) {
const auto& runtime_devices = plan.runtimes[ci].devices;
return std::find(runtime_devices.begin(), runtime_devices.end(), di) != runtime_devices.end();
};
auto headroom_for = [&](size_t ci, size_t di) {
// Higher-priority offloaded weights need cache space on their compute devices.
int64_t headroom = 0;
for (size_t other = 0; other < components.size(); ++other) {
if (components[other].params_bytes == 0 || !uses_device(other, di)) {
continue;
}
const bool resident = other == ci || plan.decisions[other].params_location == ParamsLocation::MAIN_GPU;
const int64_t cached_weights = components[other].kind < components[ci].kind
? components[other].params_bytes
: components[other].staging_bytes;
headroom = std::max(headroom, components[other].reserve_bytes +
(resident ? 0 : cached_weights));
}
return headroom;
};
for (size_t ci : order) {
const Component& comp = components[ci];
Decision& decision = plan.decisions[ci];
@@ -219,24 +260,19 @@ namespace sd::backend_fit {
continue;
}
// Higher-priority offloaded weights need GPU cache space across graph runs.
int64_t headroom = 0;
for (size_t other = 0; other < components.size(); ++other) {
if (components[other].params_bytes == 0) {
continue;
}
const bool resident = other == ci || plan.decisions[other].params_location == ParamsLocation::MAIN_GPU;
const int64_t cached_weights = components[other].kind < comp.kind
? components[other].params_bytes
: components[other].staging_bytes;
headroom = std::max(headroom, components[other].reserve_bytes +
(resident ? 0 : cached_weights));
}
int64_t& main_remaining = remaining[plan.main_device];
if (headroom <= main_remaining && comp.params_bytes <= main_remaining - headroom) {
const auto& runtime_devices = plan.runtimes[ci].devices;
const bool fits_runtime = !runtime_devices.empty() &&
std::all_of(runtime_devices.begin(), runtime_devices.end(), [&](size_t di) {
const int64_t headroom = headroom_for(ci, di);
return headroom <= remaining[di] && comp.params_bytes <= remaining[di] - headroom;
});
if (fits_runtime) {
decision.params_location = ParamsLocation::MAIN_GPU;
decision.params_device = plan.main_device;
main_remaining -= comp.params_bytes;
decision.params_device = runtime_devices.front();
// Exact split allocations are unavailable until the runners build their plans.
for (size_t di : runtime_devices) {
remaining[di] -= comp.params_bytes;
}
continue;
}
if (comp.params_bytes <= ram_budget_bytes) {
@@ -244,10 +280,14 @@ namespace sd::backend_fit {
ram_budget_bytes -= comp.params_bytes;
continue;
}
if (runtime_devices.empty()) {
continue;
}
size_t best = SIZE_MAX;
for (size_t di = 0; di < devices.size(); ++di) {
if (di != plan.main_device && comp.params_bytes <= remaining[di] &&
const int64_t headroom = headroom_for(ci, di);
if (!uses_device(ci, di) && headroom <= remaining[di] && comp.params_bytes <= remaining[di] - headroom &&
(best == SIZE_MAX || remaining[di] > remaining[best])) {
best = di;
}
@@ -280,7 +320,7 @@ namespace sd::backend_fit {
const std::vector<Device>& devices,
int64_t free_ram,
int64_t ram_budget) {
LOG_INFO("auto-fit plan (single-GPU compute on %s):", devices[plan.main_device].name.c_str());
LOG_INFO("auto-fit plan:");
LOG_INFO(" devices:");
for (const Device& device : devices) {
LOG_INFO(" %-12s %-32s free %6lld MiB, budget %6lld MiB",
@@ -293,17 +333,19 @@ namespace sd::backend_fit {
LOG_INFO(" RAM free %6lld MiB, params budget %6lld MiB",
(long long)(free_ram / MiB), (long long)(ram_budget / MiB));
}
LOG_INFO(" main-GPU weight cache priority: diffusion > te > vae");
LOG_INFO(" components (params: main GPU -> RAM -> other GPU -> disk):");
LOG_INFO(" compute-device weight cache priority: diffusion > te > vae");
LOG_INFO(" components (params: compute device -> RAM -> other GPU -> disk):");
for (size_t ci = 0; ci < components.size(); ++ci) {
const Component& comp = components[ci];
if (comp.params_bytes == 0) {
continue;
}
const std::string params = params_backend_name(plan.decisions[ci], devices);
const std::string params = plan.decisions[ci].params_location == ParamsLocation::MAIN_GPU
? plan.runtimes[ci].name
: params_backend_name(plan.decisions[ci], devices);
LOG_INFO(" %-12s params %6lld MiB, compute reserve %5lld MiB -> compute %s, params %s",
comp.name, (long long)(comp.params_bytes / MiB), (long long)(comp.reserve_bytes / MiB),
devices[plan.main_device].name.c_str(), params.c_str());
plan.runtimes[ci].name.c_str(), params.c_str());
}
}
@@ -328,6 +370,51 @@ namespace sd::backend_fit {
return "";
}
static bool resolve_runtimes(const std::vector<Component>& components,
const std::vector<Device>& devices,
std::string& runtime_spec,
std::vector<Runtime>& runtimes,
std::string& error) {
SDBackendAssignment assignment;
if (!sd_parse_backend_assignment(runtime_spec, &assignment, &error)) {
return false;
}
const size_t main_device = select_main_device(devices);
const SDBackendModule modules[] = {SDBackendModule::DIFFUSION, SDBackendModule::TE, SDBackendModule::VAE};
for (const Component& comp : components) {
std::string name = assignment.get(modules[int(comp.kind)]);
if (name.empty()) {
name = main_device == SIZE_MAX ? "cpu" : devices[main_device].name;
if (comp.params_bytes > 0) {
append_assignment(runtime_spec, module_key(comp.kind), name);
}
}
Runtime runtime;
for (const std::string& part : split_string(name, '&')) {
if (trim(part).empty()) {
continue;
}
const std::string resolved = sd_backend_resolve_name(part);
if (resolved.empty()) {
error = "backend '" + part + "' was not found";
return false;
}
if (!runtime.name.empty()) {
runtime.name += "&";
}
runtime.name += resolved;
for (size_t di = 0; di < devices.size(); ++di) {
if (devices[di].name == resolved &&
std::find(runtime.devices.begin(), runtime.devices.end(), di) == runtime.devices.end()) {
runtime.devices.push_back(di);
}
}
}
runtimes.push_back(std::move(runtime));
}
return true;
}
bool derive_backend_specs(ModelLoader& loader,
ggml_type override_wtype,
sd::ggml_graph_cut::MaxVramAssignment& budgets,
@@ -339,12 +426,18 @@ namespace sd::backend_fit {
return false;
}
const auto components = estimate_components(loader, override_wtype);
const auto devices = enumerate_gpu_devices(budgets);
// Resolve once to ensure dynamic backends are loaded before enumerating devices.
sd_backend_resolve_name("");
const auto components = estimate_components(loader, override_wtype);
const auto devices = enumerate_gpu_devices(budgets, !runtime_spec.empty());
std::vector<Runtime> runtimes;
if (!runtime_spec.empty() && !resolve_runtimes(components, devices, runtime_spec, runtimes, error)) {
LOG_ERROR("%s", error.c_str());
return false;
}
const int64_t free_ram = available_ram_bytes();
const int64_t ram_budget = std::max<int64_t>(free_ram - std::max<int64_t>(2048 * MiB, free_ram / 10), 0);
const auto plan = compute_plan(components, devices, ram_budget);
runtime_spec.clear();
const auto plan = compute_plan(components, devices, ram_budget, runtimes);
params_spec.clear();
if (!plan.valid) {
if (devices.empty()) {
@@ -362,7 +455,9 @@ namespace sd::backend_fit {
continue;
}
const char* key = module_key(components[ci].kind);
append_assignment(runtime_spec, key, devices[plan.main_device].name);
if (runtimes.empty()) {
append_assignment(runtime_spec, key, plan.runtimes[ci].name);
}
if (plan.decisions[ci].params_location != ParamsLocation::MAIN_GPU) {
append_assignment(params_spec, key, params_backend_name(plan.decisions[ci], devices));
}
@@ -389,7 +484,9 @@ namespace sd::backend_fit {
tiling_params.temporal_tiling = true;
retry_mode = tiling_params.enabled ? "spatial+temporal" : "temporal";
} else if (!tiling_params.enabled) {
tiling_params.enabled = true;
tiling_params.enabled = true;
tiling_params.rel_size_x = 0.5f;
tiling_params.rel_size_y = 0.5f;
if (tiling_params.tile_size_x <= 0) {
tiling_params.tile_size_x = 256;
}
@@ -401,7 +498,7 @@ namespace sd::backend_fit {
return false;
}
LOG_WARN("auto-fit: VAE decode failed (likely out of memory); retrying with %s tiling",
LOG_WARN("VAE decode failed (likely out of memory); retrying with %s tiling",
retry_mode);
return true;
}
+19 -5
View File
@@ -2,14 +2,16 @@
#include <algorithm>
#include <cstring>
#include <exception>
#include <map>
#include <unordered_map>
#include <unordered_set>
#include "core/ggml_extend_backend.h"
#include "core/ggml_graph_cut.h"
#include "core/util.h"
#include "ggml-cpu.h"
#include "ggml/src/ggml-impl.h"
#include "ggml-impl.h"
namespace sd {
ComputeWorkspace::~ComputeWorkspace() {
@@ -228,11 +230,23 @@ namespace sd {
}
}
void ComputeWorkspace::segment_end() {
if (active_) {
synchronize();
active_ = false;
bool ComputeWorkspace::segment_end() noexcept {
if (!active_) {
return true;
}
// Outer cleanup guards must not retry a failed backend submission.
active_ = false;
try {
synchronize();
return true;
} catch (const std::exception& error) {
LOG_ERROR("%s workspace synchronization failed during segment cleanup: %s",
ggml_backend_name(backend_), error.what());
} catch (...) {
LOG_ERROR("%s workspace synchronization failed during segment cleanup: unknown exception",
ggml_backend_name(backend_));
}
return false;
}
bool ComputeWorkspace::release() {
+1 -1
View File
@@ -51,7 +51,7 @@ namespace sd {
const std::function<ggml_backend_t(const ggml_tensor*)>& external_backend,
const AssignNodes& assign_nodes);
void synchronize() const;
void segment_end();
bool segment_end() noexcept;
bool release();
bool active() const { return active_; }
ggml_backend_sched_t scheduler() const { return scheduler_; }
+106 -6
View File
@@ -1,6 +1,7 @@
#include "core/ggml_extend.h"
#include <cmath>
#include <stdexcept>
#include <utility>
#include "core/ggml_extend_backend.h"
@@ -247,6 +248,7 @@ ggml_tensor* ggml_ext_linear_i8_tensorwise(ggml_context* ctx,
ggml_tensor* b,
int convrot_group_size,
float scale) {
#ifndef SD_USE_UPSTREAM_GGML
GGML_ASSERT(x->type == GGML_TYPE_F32 || (x->type == GGML_TYPE_I8 && scale == 1.f));
if (scale != 1.f) {
x = ggml_ext_scale(ctx, x, scale);
@@ -270,6 +272,16 @@ ggml_tensor* ggml_ext_linear_i8_tensorwise(ggml_context* ctx,
}
}
return x;
#else
GGML_UNUSED(ctx);
GGML_UNUSED(x);
GGML_UNUSED(w);
GGML_UNUSED(weight_scale);
GGML_UNUSED(b);
GGML_UNUSED(convrot_group_size);
GGML_UNUSED(scale);
throw std::runtime_error("INT8 tensorwise/convrot is not supported by this ggml build");
#endif
}
ggml_tensor* ggml_ext_pad_ext(ggml_context* ctx,
@@ -325,6 +337,76 @@ ggml_tensor* ggml_ext_pad(ggml_context* ctx,
return ggml_ext_pad_ext(ctx, nullptr, x, 0, p0, 0, p1, 0, p2, 0, p3, circular_x, circular_y);
}
static ggml_tensor* conv_1d(ggml_context* ctx, ggml_tensor* x, ggml_tensor* w, int s0, int p0, int d0, bool force_prec_f32) {
ggml_tensor* result;
if (force_prec_f32) {
ggml_tensor* patches = ggml_im2col(ctx, w, x, s0, 0, p0, 0, d0, 0, false, GGML_TYPE_F32);
result = ggml_mul_mat(ctx,
ggml_reshape_2d(ctx, patches, patches->ne[0], patches->ne[2] * patches->ne[1]),
ggml_reshape_2d(ctx, w, w->ne[0] * w->ne[1], w->ne[2]));
result = ggml_reshape_3d(ctx, result, patches->ne[1], w->ne[2], patches->ne[2]);
} else {
result = ggml_conv_1d(ctx, w, x, s0, p0, d0);
}
if (x->ne[2] > 1) {
// mul_mat packs positions and batches before output channels: [OL, N, OC].
result = ggml_reshape_3d(ctx, result, result->ne[0], x->ne[2], w->ne[2]);
result = ggml_cont(ctx, ggml_permute(ctx, result, 0, 2, 1, 3));
}
return result;
}
ggml_tensor* ggml_ext_conv_1d(ggml_context* ctx,
ggml_tensor* x,
ggml_tensor* w,
ggml_tensor* b,
int s0,
int p0,
int d0,
int64_t groups,
bool force_prec_f32) {
GGML_ASSERT(s0 > 0 && p0 >= 0 && d0 > 0 && groups > 0);
GGML_ASSERT(x->type == GGML_TYPE_F32 && x->ne[3] == 1 && w->ne[3] == 1);
GGML_ASSERT(x->ne[1] % groups == 0 && w->ne[2] % groups == 0);
GGML_ASSERT(w->ne[1] == x->ne[1] / groups);
GGML_ASSERT(b == nullptr || (b->type == GGML_TYPE_F32 && ggml_is_vector(b) && b->ne[0] == w->ne[2]));
// im2col requires contiguous time rows; group views must retain the real channel and batch strides.
if (!ggml_is_contiguous(x)) {
x = ggml_cont(ctx, x);
}
if (force_prec_f32 && w->type != GGML_TYPE_F32) {
w = ggml_cast(ctx, w, GGML_TYPE_F32);
}
if (!ggml_is_contiguous(w)) {
w = ggml_cont(ctx, w);
}
ggml_tensor* result = nullptr;
if (groups == 1) {
result = conv_1d(ctx, x, w, s0, p0, d0, force_prec_f32);
} else {
const int64_t ic_g = x->ne[1] / groups;
const int64_t oc_g = w->ne[2] / groups;
std::vector<ggml_tensor*> outputs;
outputs.reserve(groups);
for (int64_t group = 0; group < groups; ++group) {
ggml_tensor* x_i = ggml_view_3d(ctx, x, x->ne[0], ic_g, x->ne[2], x->nb[1], x->nb[2], group * ic_g * x->nb[1]);
ggml_tensor* w_i = ggml_view_3d(ctx, w, w->ne[0], ic_g, oc_g, w->nb[1], w->nb[2], group * oc_g * w->nb[2]);
outputs.push_back(conv_1d(ctx, x_i, w_i, s0, p0, d0, force_prec_f32));
}
result = ggml_ext_vec_concat(ctx, outputs, 1);
}
if (b != nullptr) {
if (!ggml_is_contiguous(b)) {
b = ggml_cont(ctx, b);
}
b = ggml_reshape_3d(ctx, b, 1, w->ne[2], 1);
result = ggml_add_inplace(ctx, result, b);
}
return result;
}
ggml_tensor* ggml_ext_conv_2d(ggml_context* ctx,
ggml_tensor* x,
ggml_tensor* w,
@@ -382,8 +464,13 @@ ggml_tensor* ggml_ext_conv_3d(ggml_context* ctx,
int d0,
int d1,
int d2,
bool force_prec_f32) {
if (force_prec_f32) {
bool force_prec_f32,
bool direct) {
if (direct) {
int64_t OC = w->ne[3] / IC;
int64_t N = x->ne[3] / IC;
x = ggml_conv_3d_direct(ctx, w, x, s0, s1, s2, p0, p1, p2, d0, d1, d2, (int)IC, (int)N, (int)OC);
} else if (force_prec_f32) {
ggml_tensor* im2col = ggml_im2col_3d(ctx, w, x, IC, s0, s1, s2, p0, p1, p2, d0, d1, d2, w->type);
int64_t OC = w->ne[3] / IC;
@@ -573,6 +660,14 @@ ggml_tensor* ggml_ext_attention_ext(ggml_context* ctx,
ggml_tensor* kqv = nullptr;
auto build_kqv = [&](ggml_tensor* q_in, ggml_tensor* k_in, ggml_tensor* v_in, ggml_tensor* mask_in) -> ggml_tensor* {
const bool pad_head = d_head > 0 && d_head < 64 && q_in->ne[0] == d_head && k_in->ne[0] == d_head &&
q_in->type == GGML_TYPE_F32 && k_in->type == GGML_TYPE_F32 &&
v_in->type == GGML_TYPE_F32 && sd_backend_supports_cuda_mma(backend);
if (pad_head) {
// CUDA FA MMA starts at 64 channels; keep the original head's attention scale.
q_in = ggml_pad(ctx, q_in, 64 - d_head, 0, 0, 0);
k_in = ggml_pad(ctx, k_in, 64 - d_head, 0, 0, 0);
}
if (kv_scale != 1.0f) {
k_in = ggml_ext_scale(ctx, k_in, kv_scale);
}
@@ -580,6 +675,9 @@ ggml_tensor* ggml_ext_attention_ext(ggml_context* ctx,
v_in = ggml_ext_cont(ctx, ggml_permute(ctx, v_in, 0, 2, 1, 3));
v_in = ggml_reshape_3d(ctx, v_in, d_head, L_k, n_kv_head * N);
if (pad_head) {
v_in = ggml_pad(ctx, v_in, 64 - d_head, 0, 0, 0);
}
if (kv_scale != 1.0f) {
v_in = ggml_ext_scale(ctx, v_in, kv_scale);
}
@@ -609,6 +707,9 @@ ggml_tensor* ggml_ext_attention_ext(ggml_context* ctx,
if (kv_scale != 1.0f) {
out = ggml_ext_scale(ctx, out, 1.0f / kv_scale);
}
if (pad_head) {
out = ggml_ext_slice(ctx, out, 0, 0, d_head);
}
return out;
};
@@ -683,17 +784,16 @@ ggml_tensor* ggml_ext_group_norm(ggml_context* ctx,
ggml_tensor* x,
ggml_tensor* w,
ggml_tensor* b,
int num_groups) {
int num_groups,
float eps) {
if (ggml_n_dims(x) >= 3 && w != nullptr && b != nullptr) {
w = ggml_reshape_4d(ctx, w, 1, 1, w->ne[0], 1);
b = ggml_reshape_4d(ctx, b, 1, 1, b->ne[0], 1);
}
const float eps = 1e-6f; // default eps parameter
x = ggml_group_norm(ctx, x, num_groups, eps);
x = ggml_group_norm(ctx, x, num_groups, eps);
if (w != nullptr && b != nullptr) {
x = ggml_mul_inplace(ctx, x, w);
// b = ggml_repeat(ctx, b, x);
x = ggml_add_inplace(ctx, x, b);
}
return x;
+17 -3
View File
@@ -9,7 +9,7 @@
#define EPS 1e-05f
static_assert(GGML_MAX_NAME >= 128, "GGML_MAX_NAME must be at least 128");
static_assert(GGML_MAX_NAME >= 160, "GGML_MAX_NAME must be at least 160");
// n-mode tensor-matrix product
// example: 2-mode product
@@ -103,6 +103,18 @@ ggml_tensor* ggml_ext_pad(ggml_context* ctx,
bool circular_x = false,
bool circular_y = false);
// ggml layout: x [L, IC, N], w [K, IC/groups, OC], b [OC], result [OL, OC, N].
// force_prec_f32 keeps both input patches and weights in F32.
ggml_tensor* ggml_ext_conv_1d(ggml_context* ctx,
ggml_tensor* x,
ggml_tensor* w,
ggml_tensor* b,
int s0 = 1,
int p0 = 0,
int d0 = 1,
int64_t groups = 1,
bool force_prec_f32 = false);
// w: [OCIC, KH, KW]
// x: [N, IC, IH, IW]
// b: [OC,]
@@ -141,7 +153,8 @@ ggml_tensor* ggml_ext_conv_3d(ggml_context* ctx,
int d0 = 1,
int d1 = 1,
int d2 = 1,
bool force_prec_f32 = false);
bool force_prec_f32 = false,
bool direct = false);
// w: [OCIC, KD, 1 * 1]
// x: [N, IC, ID, IH*IW]
@@ -219,7 +232,8 @@ ggml_tensor* ggml_ext_group_norm(ggml_context* ctx,
ggml_tensor* x,
ggml_tensor* w,
ggml_tensor* b,
int num_groups = 32);
int num_groups = 32,
float eps = 1e-6f);
ggml_tensor* ggml_ext_timestep_embedding(
ggml_context* ctx,
+83 -3
View File
@@ -8,8 +8,12 @@
#include <stdexcept>
#include <vector>
#ifdef SD_USE_CUDA
#include <cuda.h>
#endif
#include "core/util.h"
#include "ggml/src/ggml-impl.h"
#include "ggml-impl.h"
#include "stable-diffusion.h"
static std::string trim_copy(const std::string& value) {
@@ -87,6 +91,10 @@ static bool parse_backend_module(const std::string& raw_name, SDBackendModule* m
*module = SDBackendModule::DETECTOR;
return true;
}
if (name == "audioencoder" || name == "audio") {
*module = SDBackendModule::AUDIO_ENCODER;
return true;
}
return false;
}
@@ -425,6 +433,70 @@ bool sd_backend_is_cpu(ggml_backend_t backend) {
return dev != nullptr && ggml_backend_dev_type(dev) == GGML_BACKEND_DEVICE_TYPE_CPU;
}
bool sd_backend_supports_cuda_mma(ggml_backend_t backend) {
#ifdef SD_USE_CUDA
if (!sd_backend_is(backend, "CUDA")) {
return false;
}
auto dev = ggml_backend_get_device(backend);
if (dev == nullptr) {
return false;
}
static std::mutex mutex;
static std::unordered_map<ggml_backend_dev_t, bool> cache;
std::lock_guard<std::mutex> lock(mutex);
auto it = cache.find(dev);
if (it != cache.end()) {
return it->second;
}
const bool supported = [&]() {
ggml_backend_dev_props props{};
ggml_backend_dev_get_props(dev, &props);
CUdevice device;
int major = 0, minor = 0;
if (props.device_id == nullptr || cuInit(0) != CUDA_SUCCESS ||
cuDeviceGetByPCIBusId(&device, props.device_id) != CUDA_SUCCESS ||
cuDeviceGetAttribute(&major, CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MAJOR, device) != CUDA_SUCCESS ||
cuDeviceGetAttribute(&minor, CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MINOR, device) != CUDA_SUCCESS) {
return false;
}
auto reg = ggml_backend_dev_backend_reg(dev);
auto get_features = reinterpret_cast<ggml_backend_get_features_t>(
ggml_backend_reg_get_proc_address(reg, "ggml_backend_get_features"));
if (get_features == nullptr) {
return false;
}
// Match ggml's highest compiled architecture for this device, including PTX fallback.
const int cc = 100 * major + 10 * minor;
int compiled_arch = 0;
for (auto feature = get_features(reg); feature != nullptr && feature->name != nullptr; ++feature) {
if (std::strcmp(feature->name, "ARCHS") != 0 || feature->value == nullptr) {
continue;
}
const char* arch = feature->value;
while (*arch != '\0') {
char* end = nullptr;
const long value = std::strtol(arch, &end, 10);
if (end == arch) {
++arch;
continue;
}
if (value <= cc && value > compiled_arch) {
compiled_arch = static_cast<int>(value);
}
arch = end;
}
}
return compiled_arch == 700 || compiled_arch >= 750;
}();
cache.emplace(dev, supported);
return supported;
#else
(void)backend;
return false;
#endif
}
ggml_backend_t sd_backend_cpu_init() {
ggml_backend_load_all_once();
return ggml_backend_init_by_type(GGML_BACKEND_DEVICE_TYPE_CPU, nullptr);
@@ -593,7 +665,7 @@ static ggml_backend_t sd_get_default_backend() {
return backend;
}
static bool sd_parse_backend_assignment(const std::string& spec, SDBackendAssignment* assignment, std::string* error) {
bool sd_parse_backend_assignment(const std::string& spec, SDBackendAssignment* assignment, std::string* error) {
if (assignment == nullptr) {
return false;
}
@@ -660,7 +732,13 @@ void SDBackendAssignment::set_module(SDBackendModule module, const std::string&
}
void SDBackendHandleDeleter::operator()(ggml_backend_t backend) const {
ggml_backend_free(backend);
try {
ggml_backend_free(backend);
} catch (const std::exception& error) {
LOG_ERROR("backend cleanup failed: %s", error.what());
} catch (...) {
LOG_ERROR("backend cleanup failed: unknown exception");
}
}
SDBackendManager::~SDBackendManager() {
@@ -962,6 +1040,8 @@ const char* sd_backend_module_name(SDBackendModule module) {
return "upscaler";
case SDBackendModule::DETECTOR:
return "detector";
case SDBackendModule::AUDIO_ENCODER:
return "audio_encoder";
}
return "unknown";
}
+3
View File
@@ -21,6 +21,7 @@ enum class SDBackendModule {
PHOTOMAKER,
UPSCALER,
DETECTOR,
AUDIO_ENCODER,
};
struct SDBackendAssignment {
@@ -86,6 +87,7 @@ private:
bool sd_backend_is(ggml_backend_t backend, const std::string& name);
bool sd_backend_is_cpu(ggml_backend_t backend);
bool sd_backend_supports_cuda_mma(ggml_backend_t backend);
ggml_backend_t sd_backend_cpu_init();
bool sd_backend_cpu_set_n_threads(ggml_backend_t backend_cpu, int n_threads);
ggml_status sd_backend_graph_compute_with_eval_callback(ggml_backend_t backend,
@@ -93,6 +95,7 @@ ggml_status sd_backend_graph_compute_with_eval_callback(ggml_backend_t backend,
sd_graph_eval_callback_t callback_eval,
void* callback_eval_user_data);
std::string sd_backend_resolve_name(const std::string& name);
bool sd_parse_backend_assignment(const std::string& spec, SDBackendAssignment* assignment, std::string* error);
const char* sd_backend_module_name(SDBackendModule module);
void ggml_ext_im_set_f32_1d(const struct ggml_tensor* tensor, int i, float value);
bool add_rpc_devices(const std::string& servers);
+100 -31
View File
@@ -16,7 +16,7 @@
#include "ggml-alloc.h"
#include "ggml-backend.h"
#include "ggml/src/ggml-impl.h"
#include "ggml-impl.h"
namespace sd::ggml_graph_cut {
@@ -426,8 +426,8 @@ namespace sd::ggml_graph_cut {
if (tensor == nullptr || tensor->name[0] == '\0') {
return false;
}
return starts_with(tensor->name, GGML_RUNNER_CUT_PREFIX) &&
ends_with(tensor->name, GGML_RUNNER_CUT_SUFFIX);
return std::strncmp(tensor->name, GGML_RUNNER_CUT_PREFIX, std::strlen(GGML_RUNNER_CUT_PREFIX)) == 0 &&
tensor->name[std::strlen(tensor->name) - 1] == GGML_RUNNER_CUT_SUFFIX[0];
}
std::string make_graph_cut_name(const std::string& group, const std::string& output) {
@@ -482,35 +482,98 @@ namespace sd::ggml_graph_cut {
return ggml_nbytes(cache_src);
}
std::vector<uint64_t> graph_layout(ggml_cgraph* graph, bool include_bindings) {
std::vector<const ggml_tensor*> tensors;
std::unordered_map<const ggml_tensor*, size_t> indices;
auto add = [&](const ggml_tensor* tensor) {
if (tensor != nullptr && indices.emplace(tensor, tensors.size() + 1).second) {
tensors.push_back(tensor);
}
static bool can_ignore_op_params(ggml_op op) {
// Exempt only parameters that cannot affect graph layout or backend allocation size.
switch (op) {
case GGML_OP_SCALE:
return true;
default:
return false;
}
}
struct GraphLayoutTensors {
struct Entry {
const ggml_tensor* tensor = nullptr;
size_t index = 0;
};
std::vector<ggml_tensor*> tensors;
std::vector<Entry> entries;
explicit GraphLayoutTensors(size_t graph_size) {
tensors.reserve(graph_size);
size_t capacity = 2;
while (capacity < 2 * graph_size) {
capacity *= 2;
}
entries.resize(capacity);
}
size_t find(const ggml_tensor* tensor) const {
size_t hash = reinterpret_cast<uintptr_t>(tensor) >> 4;
hash ^= hash >> 16;
const size_t mask = entries.size() - 1;
size_t slot = hash & mask;
while (entries[slot].tensor != nullptr && entries[slot].tensor != tensor) {
slot = (slot + 1) & mask;
}
return slot;
}
void add(ggml_tensor* tensor) {
if (tensor == nullptr) {
return;
}
size_t slot = find(tensor);
if (entries[slot].tensor != nullptr) {
return;
}
if (2 * (tensors.size() + 1) > entries.size()) {
// Segment graphs can reference tensors outside their node and leaf arrays.
std::vector<Entry> next(2 * entries.size());
entries.swap(next);
for (size_t i = 0; i < tensors.size(); ++i) {
entries[find(tensors[i])] = {tensors[i], i + 1};
}
slot = find(tensor);
}
entries[slot] = {tensor, tensors.size() + 1};
tensors.push_back(tensor);
}
size_t index(const ggml_tensor* tensor) const {
return tensor == nullptr ? 0 : entries[find(tensor)].index;
}
};
std::vector<uint64_t> graph_layout(ggml_cgraph* graph, bool include_bindings) {
const size_t graph_size = static_cast<size_t>(graph->n_leafs) + graph->n_nodes;
GraphLayoutTensors layout_tensors(graph_size);
const auto& tensors = layout_tensors.tensors;
for (int i = 0; i < graph->n_leafs; ++i) {
add(graph->leafs[i]);
layout_tensors.add(graph->leafs[i]);
}
for (int i = 0; i < graph->n_nodes; ++i) {
add(graph->nodes[i]);
layout_tensors.add(graph->nodes[i]);
}
for (size_t i = 0; i < tensors.size(); ++i) {
add(tensors[i]->view_src);
layout_tensors.add(tensors[i]->view_src);
for (auto source : tensors[i]->src) {
add(source);
layout_tensors.add(source);
}
}
std::vector<uint64_t> signature;
signature.reserve(tensors.size() * 24);
const size_t tensor_fields = 5 + 2 * GGML_MAX_DIMS + GGML_MAX_SRC +
GGML_MAX_OP_PARAMS / sizeof(int32_t) + (include_bindings ? 2 : 0);
signature.reserve(2 + graph_size + tensors.size() * tensor_fields);
signature.push_back(graph->n_nodes);
signature.push_back(graph->n_leafs);
for (int i = 0; i < graph->n_leafs; ++i) {
signature.push_back(indices.at(graph->leafs[i]));
signature.push_back(layout_tensors.index(graph->leafs[i]));
}
for (int i = 0; i < graph->n_nodes; ++i) {
signature.push_back(indices.at(graph->nodes[i]));
signature.push_back(layout_tensors.index(graph->nodes[i]));
}
for (auto tensor : tensors) {
signature.push_back(tensor->op);
@@ -526,12 +589,14 @@ namespace sd::ggml_graph_cut {
signature.push_back(tensor->ne[d]);
signature.push_back(tensor->nb[d]);
}
signature.push_back(tensor->view_src == nullptr ? 0 : indices.at(tensor->view_src));
signature.push_back(layout_tensors.index(tensor->view_src));
for (auto source : tensor->src) {
signature.push_back(source == nullptr ? 0 : indices.at(source));
signature.push_back(layout_tensors.index(source));
}
for (int value : tensor->op_params) {
signature.push_back(static_cast<uint32_t>(value));
if (!can_ignore_op_params(tensor->op)) {
for (int value : tensor->op_params) {
signature.push_back(static_cast<uint32_t>(value));
}
}
}
return signature;
@@ -550,14 +615,18 @@ namespace sd::ggml_graph_cut {
return false;
}
}
std::vector<std::pair<int, std::string>> cut_markers;
for (int i = 0; i < ggml_graph_n_nodes(gf); ++i) {
auto node = ggml_graph_node(gf, i);
size_t cut_index = 0;
for (int i = 0; i < gf->n_nodes; ++i) {
auto node = gf->nodes[i];
if (is_graph_cut_tensor(node)) {
cut_markers.emplace_back(i, node->name);
if (cut_index >= plan.cut_markers.size() ||
plan.cut_markers[cut_index].first != i || plan.cut_markers[cut_index].second != node->name) {
return false;
}
++cut_index;
}
}
return cut_markers == plan.cut_markers;
return cut_index == plan.cut_markers.size();
}
bool plan_matches_graph(ggml_cgraph* gf, const Plan& plan) {
@@ -936,11 +1005,11 @@ namespace sd::ggml_graph_cut {
return plan;
}
Plan resolve_plan(ggml_backend_t backend,
ggml_cgraph* gf,
PlanCache* cache,
const std::unordered_set<const ggml_tensor*>& params_tensor_set,
const char* log_desc) {
const Plan& resolve_plan(ggml_backend_t backend,
ggml_cgraph* gf,
PlanCache* cache,
const std::unordered_set<const ggml_tensor*>& params_tensor_set,
const char* log_desc) {
GGML_ASSERT(backend != nullptr);
GGML_ASSERT(gf != nullptr);
GGML_ASSERT(cache != nullptr);
+6 -5
View File
@@ -94,11 +94,12 @@ namespace sd::ggml_graph_cut {
ggml_cgraph* gf,
const std::unordered_set<const ggml_tensor*>& params_tensor_set,
const char* log_desc);
Plan resolve_plan(ggml_backend_t backend,
ggml_cgraph* gf,
PlanCache* cache,
const std::unordered_set<const ggml_tensor*>& params_tensor_set,
const char* log_desc);
// The returned reference is valid until its cache entry is evicted or the cache is destroyed.
const Plan& resolve_plan(ggml_backend_t backend,
ggml_cgraph* gf,
PlanCache* cache,
const std::unordered_set<const ggml_tensor*>& params_tensor_set,
const char* log_desc);
} // namespace sd::ggml_graph_cut
+56 -29
View File
@@ -1,7 +1,9 @@
#include <algorithm>
#include <exception>
#include <map>
#include <utility>
#include "core/ggml_extend.h"
#include "core/ggml_extend_backend.h"
#include "core/ggml_runner.h"
#include "core/ggml_tensor_utils.h"
@@ -11,6 +13,21 @@
using namespace sd;
ggml_tensor* ggml_ext_attention_ext(GGMLRunnerContext* ctx,
ggml_tensor* q,
ggml_tensor* k,
ggml_tensor* v,
int64_t n_head,
ggml_tensor* mask,
bool skip_reshape,
bool flash_attn,
float kv_scale) {
if (ctx->attn_scale > 0.f) {
kv_scale = ctx->attn_scale;
}
return ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, q, k, v, n_head, mask, skip_reshape, flash_attn, kv_scale);
}
void GGMLRunner::alloc_params_ctx() {
ggml_init_params params;
params.mem_size = static_cast<size_t>(MAX_PARAMS_TENSOR_NUM * ggml_tensor_overhead());
@@ -325,21 +342,17 @@ void GGMLRunner::copy_data_to_backend_tensor(ggml_cgraph* gf, bool clear_after_c
}
}
bool GGMLRunner::resolve_graph_cut_plan(ggml_cgraph* gf,
GraphCutPlan* plan_out) {
GGML_ASSERT(plan_out != nullptr);
const GGMLRunner::GraphCutPlan& GGMLRunner::resolve_graph_cut_plan(ggml_cgraph* gf) {
GGML_ASSERT(gf != nullptr);
*plan_out = sd::ggml_graph_cut::resolve_plan(runtime_backend,
gf,
&graph_cut_plan_cache_,
params_tensor_set_,
get_desc().c_str());
return true;
return sd::ggml_graph_cut::resolve_plan(runtime_backend,
gf,
&graph_cut_plan_cache_,
params_tensor_set_,
get_desc().c_str());
}
bool GGMLRunner::resolve_graph_cut_layer_split_plan(ggml_cgraph* gf,
GraphCutPlan* plan_out) {
return resolve_graph_cut_plan(gf, plan_out);
const GGMLRunner::GraphCutPlan& GGMLRunner::resolve_graph_cut_layer_split_plan(ggml_cgraph* gf) {
return resolve_graph_cut_plan(gf);
}
bool GGMLRunner::assign_graph_cut_layer_split_backends(ggml_cgraph* gf) {
@@ -352,10 +365,7 @@ bool GGMLRunner::assign_graph_cut_layer_split_backends(ggml_cgraph* gf) {
return false;
}
GraphCutPlan plan;
if (!resolve_graph_cut_layer_split_plan(gf, &plan)) {
return false;
}
const auto& plan = resolve_graph_cut_layer_split_plan(gf);
if (!plan.valid || !plan.has_cuts || plan.segments.size() <= 1) {
auto manager = residency_manager.lock();
if (manager == nullptr) {
@@ -510,7 +520,10 @@ GGMLRunnerContext GGMLRunner::get_context() {
runner_ctx.ggml_ctx = compute_ctx;
runner_ctx.backend = runtime_backend;
runner_ctx.flash_attn_enabled = flash_attn_enabled;
runner_ctx.linear_scale = linear_scale;
runner_ctx.attn_scale = attn_scale;
runner_ctx.conv2d_direct_enabled = conv2d_direct_enabled;
runner_ctx.conv3d_direct_enabled = conv3d_direct_enabled;
runner_ctx.circular_x_enabled = circular_x_enabled;
runner_ctx.circular_y_enabled = circular_y_enabled;
runner_ctx.weight_adapter = weight_adapter;
@@ -624,8 +637,15 @@ std::optional<sd::Tensor<float>> GGMLRunner::compute(get_graph_cb_t get_graph,
params_tensor_set_.insert(parameter);
}
}
auto output = execute_graph(graph, n_threads, no_return, read_outputs);
success = output.has_value();
std::optional<sd::Tensor<float>> output;
try {
output = execute_graph(graph, n_threads, no_return, read_outputs);
} catch (const std::exception& error) {
LOG_ERROR("%s graph execution failed on %s: %s", get_desc().c_str(),
ggml_backend_name(runtime_backend), error.what());
return std::nullopt;
}
success = output.has_value();
if (success) {
cache_.graph_end(true);
}
@@ -793,24 +813,25 @@ std::optional<Tensor<float>> GGMLRunner::execute_graph(ggml_cgraph* graph, int n
if (!assign_graph_cut_layer_split_backends(graph)) {
return std::nullopt;
}
const auto params = collect_used_param_tensors(graph);
ggml_graph_cut::Plan plan;
if (!resolve_graph_cut_plan(graph, &plan)) {
return std::nullopt;
}
const auto full_measurement = measure(graph, plan.compute_buffer_size);
const auto params = collect_used_param_tensors(graph);
const auto& cached_plan = resolve_graph_cut_plan(graph);
const auto full_measurement = measure(graph, cached_plan.compute_buffer_size);
if (full_measurement.buffers.empty()) {
return std::nullopt;
}
auto manager = residency_manager.lock();
const bool segmented = !is_multi_device() && !sd_backend_is_cpu(runtime_backend) &&
manager != nullptr && manager->segmented_compute_enabled() &&
plan.valid && plan.has_cuts && plan.segments.size() > 1 &&
cached_plan.valid && cached_plan.has_cuts && cached_plan.segments.size() > 1 &&
!fits(memory_requests(full_measurement.buffers, cache_.pending_bytes(graph)), params);
ggml_graph_cut::Plan monolithic_plan;
if (!segmented) {
ggml_graph_cut::Segment segment;
monolithic_plan.segments.emplace_back();
auto& segment = monolithic_plan.segments.back();
segment.group_name = "graph";
segment.compute_buffer_size = plan.compute_buffer_size;
segment.compute_buffer_size = cached_plan.compute_buffer_size;
segment.internal_node_indices.reserve(ggml_graph_n_nodes(graph));
segment.input_refs.reserve(ggml_graph_cut::leaf_count(graph));
for (int i = 0; i < ggml_graph_n_nodes(graph); ++i) {
segment.internal_node_indices.push_back(i);
}
@@ -823,8 +844,8 @@ std::optional<Tensor<float>> GGMLRunner::execute_graph(ggml_cgraph* graph, int n
: ggml_graph_cut::Segment::INPUT_EXTERNAL;
segment.input_refs.push_back(input);
}
plan.segments = {std::move(segment)};
}
const auto& plan = segmented ? cached_plan : monolithic_plan;
const bool segments_changed = plan.segments.size() != logged_segment_count_;
if (segments_changed && (segmented || logged_segment_count_ > 1)) {
LOG_VERBOSE("%s using %zu segment%s", get_desc().c_str(),
@@ -883,7 +904,10 @@ std::optional<Tensor<float>> GGMLRunner::execute_graph(ggml_cgraph* graph, int n
auto ensure_capacity = [&]() {
sync_runtime_residency();
auto requests = memory_requests(measurement.buffers, new_cache_bytes);
if (!fits(requests, weights.params(index)) && workspace_.release_excess(measurement)) {
if (fits(requests, weights.params(index))) {
return true;
}
if (workspace_.release_excess(measurement)) {
sync_runtime_residency();
requests = memory_requests(measurement.buffers, new_cache_bytes);
}
@@ -938,6 +962,9 @@ std::optional<Tensor<float>> GGMLRunner::execute_graph(ggml_cgraph* graph, int n
}
}
}
if (!workspace_.segment_end()) {
return fail_segment("workspace synchronization");
}
// Final outputs and their callbacks may still be views of consumed cuts.
cut_cache_.prune(segment.future_cut_names);
}
+27 -4
View File
@@ -68,7 +68,10 @@ struct GGMLRunnerContext {
ggml_backend_t backend = nullptr;
ggml_context* ggml_ctx = nullptr;
bool flash_attn_enabled = false;
float linear_scale = 0.f;
float attn_scale = 0.f;
bool conv2d_direct_enabled = false;
bool conv3d_direct_enabled = false;
bool circular_x_enabled = false;
bool circular_y_enabled = false;
ggml_tensor* ip_context = nullptr;
@@ -113,6 +116,16 @@ struct GGMLRunnerContext {
}
};
ggml_tensor* ggml_ext_attention_ext(GGMLRunnerContext* ctx,
ggml_tensor* q,
ggml_tensor* k,
ggml_tensor* v,
int64_t n_head,
ggml_tensor* mask = nullptr,
bool skip_reshape = false,
bool flash_attn = false,
float kv_scale = 1.f);
struct GGMLRunner {
private:
std::map<ggml_backend_t, size_t> logged_compute_bytes_;
@@ -163,7 +176,10 @@ protected:
const std::string final_result_name = "ggml_runner_final_result_tensor";
bool flash_attn_enabled = false;
float linear_scale = 0.f;
float attn_scale = 0.f;
bool conv2d_direct_enabled = false;
bool conv3d_direct_enabled = false;
bool circular_x_enabled = false;
bool circular_y_enabled = false;
@@ -249,11 +265,9 @@ protected:
void copy_data_to_backend_tensor(ggml_cgraph* gf, bool clear_after_copy = true);
bool resolve_graph_cut_plan(ggml_cgraph* gf,
GraphCutPlan* plan_out);
const GraphCutPlan& resolve_graph_cut_plan(ggml_cgraph* gf);
bool resolve_graph_cut_layer_split_plan(ggml_cgraph* gf,
GraphCutPlan* plan_out);
const GraphCutPlan& resolve_graph_cut_layer_split_plan(ggml_cgraph* gf);
bool assign_graph_cut_layer_split_backends(ggml_cgraph* gf);
@@ -323,10 +337,19 @@ public:
flash_attn_enabled = enabled;
}
void set_scale_overrides(float linear_scale, float attn_scale) {
this->linear_scale = linear_scale;
this->attn_scale = attn_scale;
}
void set_conv2d_direct_enabled(bool enabled) {
conv2d_direct_enabled = enabled;
}
void set_conv3d_direct_enabled(bool enabled) {
conv3d_direct_enabled = enabled;
}
void set_circular_axes(bool circular_x, bool circular_y) {
circular_x_enabled = circular_x;
circular_y_enabled = circular_y;
+143
View File
@@ -0,0 +1,143 @@
#include "core/parallel.h"
#include <algorithm>
#include <condition_variable>
#include <exception>
#include <mutex>
#include <thread>
#include <vector>
namespace sd {
namespace parallel_detail {
thread_local ParallelExecutor* executor = nullptr;
thread_local bool active = false;
}
struct ParallelExecutor::Impl {
struct Worker {
std::condition_variable wake;
std::thread thread;
bool ready = false;
};
ParallelExecutor* owner;
std::mutex invocation_mutex;
std::mutex mutex;
std::condition_variable finished;
std::vector<std::unique_ptr<Worker>> workers;
bool stopping = false;
int pending = 0;
int participants = 1;
int64_t begin = 0;
int64_t count = 0;
const std::function<void(int64_t, int64_t)>* task = nullptr;
std::exception_ptr error;
explicit Impl(ParallelExecutor* owner)
: owner(owner) {}
~Impl() {
{
std::lock_guard<std::mutex> lock(mutex);
stopping = true;
}
for (auto& worker : workers) {
worker->wake.notify_one();
}
for (auto& worker : workers) {
worker->thread.join();
}
}
void execute(int index) {
ParallelScope scope(owner);
parallel_detail::Region region;
const int64_t size = count / participants;
const int64_t extra = count % participants;
const int64_t first = begin + index * size + std::min<int64_t>(index, extra);
const int64_t last = first + size + (index < extra ? 1 : 0);
try {
(*task)(first, last);
} catch (...) {
std::lock_guard<std::mutex> lock(mutex);
if (!error) {
error = std::current_exception();
}
}
}
void worker_loop(Worker* worker, int index) {
std::unique_lock<std::mutex> lock(mutex);
for (;;) {
worker->wake.wait(lock, [&] { return stopping || worker->ready; });
if (stopping) {
return;
}
worker->ready = false;
lock.unlock();
execute(index);
lock.lock();
if (--pending == 0) {
finished.notify_one();
}
}
}
void run(int64_t first, int64_t last, int n_tasks, const std::function<void(int64_t, int64_t)>& callback) {
std::lock_guard<std::mutex> invocation_lock(invocation_mutex);
std::unique_lock<std::mutex> lock(mutex);
while (static_cast<int>(workers.size()) < n_tasks - 1) {
workers.push_back(std::make_unique<Worker>());
auto* worker = workers.back().get();
const int index = static_cast<int>(workers.size());
try {
worker->thread = std::thread([this, worker, index] { worker_loop(worker, index); });
} catch (...) {
workers.pop_back();
throw;
}
}
begin = first;
count = last - first;
participants = n_tasks;
pending = n_tasks - 1;
task = &callback;
error = nullptr;
for (int i = 0; i < pending; ++i) {
workers[i]->ready = true;
workers[i]->wake.notify_one();
}
lock.unlock();
execute(0);
lock.lock();
finished.wait(lock, [&] { return pending == 0; });
task = nullptr;
if (error) {
std::rethrow_exception(error);
}
}
};
ParallelExecutor::ParallelExecutor(int n_threads)
: n_threads_(std::max(1, n_threads)), impl_(std::make_unique<Impl>(this)) {}
ParallelExecutor::~ParallelExecutor() = default;
void ParallelExecutor::run(int64_t begin, int64_t end, int64_t grain_size, const std::function<void(int64_t, int64_t)>& task) {
if (begin < 0 || grain_size <= 0) {
throw std::invalid_argument("parallel_for requires begin >= 0 and grain_size > 0");
}
if (end <= begin) {
return;
}
const int n_tasks = static_cast<int>(std::min<int64_t>(n_threads_, (end - begin) / grain_size));
if (parallel_detail::active || n_tasks <= 1) {
parallel_detail::Region region;
task(begin, end);
return;
}
impl_->run(begin, end, n_tasks, task);
}
}
+77
View File
@@ -0,0 +1,77 @@
#ifndef __SD_CORE_PARALLEL_H__
#define __SD_CORE_PARALLEL_H__
#include <cstdint>
#include <functional>
#include <memory>
#include <stdexcept>
#include <utility>
namespace sd {
class ParallelExecutor {
struct Impl;
int n_threads_;
std::unique_ptr<Impl> impl_;
public:
explicit ParallelExecutor(int n_threads);
~ParallelExecutor();
ParallelExecutor(const ParallelExecutor&) = delete;
ParallelExecutor& operator=(const ParallelExecutor&) = delete;
int num_threads() const { return n_threads_; }
void run(int64_t begin, int64_t end, int64_t grain_size, const std::function<void(int64_t, int64_t)>& task);
};
namespace parallel_detail {
extern thread_local ParallelExecutor* executor;
extern thread_local bool active;
class Region {
bool previous_;
public:
Region()
: previous_(active) { active = true; }
~Region() { active = previous_; }
Region(const Region&) = delete;
Region& operator=(const Region&) = delete;
};
}
class ParallelScope {
ParallelExecutor* previous_;
public:
explicit ParallelScope(ParallelExecutor* executor)
: previous_(parallel_detail::executor) {
parallel_detail::executor = executor;
}
~ParallelScope() { parallel_detail::executor = previous_; }
ParallelScope(const ParallelScope&) = delete;
ParallelScope& operator=(const ParallelScope&) = delete;
};
// Ranges are non-negative. The callback may run concurrently and must own its writes.
template <typename F>
inline void parallel_for(int64_t begin, int64_t end, int64_t grain_size, F&& task) {
if (begin < 0 || grain_size <= 0) {
throw std::invalid_argument("parallel_for requires begin >= 0 and grain_size > 0");
}
if (end <= begin) {
return;
}
auto* executor = parallel_detail::executor;
if (parallel_detail::active || executor == nullptr || executor->num_threads() <= 1 ||
(end - begin) / grain_size < 2) {
parallel_detail::Region region;
task(begin, end);
return;
}
executor->run(begin, end, grain_size, std::forward<F>(task));
}
}
#endif // __SD_CORE_PARALLEL_H__
+171
View File
@@ -0,0 +1,171 @@
#include "regex.h"
#include <cstdint>
#include <limits>
#include <mutex>
#define ONIG_ESCAPE_UCHAR_COLLISION
#define ONIG_ESCAPE_REGEX_T_COLLISION
#include <oniguruma.h>
namespace sd {
struct Regex::Impl {
OnigRegex regex = nullptr;
~Impl() {
if (regex) {
onig_free(regex);
}
}
};
struct RegexRegionDeleter {
void operator()(OnigRegion* region) const {
onig_region_free(region, 1);
}
};
static bool regex_error(std::string* error, const std::string& message) {
if (error) {
*error = message;
}
return false;
}
static bool regex_onig_error(std::string* error, int code, OnigErrorInfo* info = nullptr) {
OnigUChar buffer[ONIG_MAX_ERROR_MESSAGE_LEN];
onig_error_code_to_str(buffer, code, info);
return regex_error(error, reinterpret_cast<const char*>(buffer));
}
static int regex_initialize() {
static std::once_flag once;
static int result = ONIG_NORMAL;
std::call_once(once, [] {
OnigEncoding encodings[] = {ONIG_ENCODING_UTF8};
result = onig_initialize(encodings, 1);
});
// onig_end() would invalidate expressions held by other Regex instances.
return result;
}
// Rust str excludes overlong encodings, surrogates and extended UTF-8 accepted by Oniguruma.
static bool regex_valid_utf8(const std::string& text) {
size_t position = 0;
while (position < text.size()) {
const auto lead = static_cast<unsigned char>(text[position++]);
if (lead < 0x80) {
continue;
}
int count = 0;
if (lead >= 0xC2 && lead <= 0xDF) {
count = 1;
} else if (lead >= 0xE0 && lead <= 0xEF) {
count = 2;
} else if (lead >= 0xF0 && lead <= 0xF4) {
count = 3;
}
if (count == 0 || text.size() - position < static_cast<size_t>(count)) {
return false;
}
uint32_t codepoint = lead & (0x7F >> count);
for (int i = 0; i < count; ++i) {
const auto byte = static_cast<unsigned char>(text[position++]);
if ((byte & 0xC0) != 0x80) {
return false;
}
codepoint = (codepoint << 6) | (byte & 0x3F);
}
constexpr uint32_t minimum[] = {0, 0x80, 0x800, 0x10000};
if (codepoint < minimum[count] || codepoint > 0x10FFFF || (codepoint >= 0xD800 && codepoint <= 0xDFFF)) {
return false;
}
}
return true;
}
Regex::Regex() = default;
Regex::~Regex() = default;
Regex::Regex(Regex&&) noexcept = default;
Regex& Regex::operator=(Regex&&) noexcept = default;
bool Regex::compile(const std::string& pattern, std::string* error) {
if (error) {
error->clear();
}
const int initialized = regex_initialize();
if (initialized != ONIG_NORMAL) {
return regex_onig_error(error, initialized);
}
if (pattern.size() > static_cast<size_t>(std::numeric_limits<int>::max())) {
return regex_error(error, "regex pattern exceeds Oniguruma's offset range");
}
const auto* begin = reinterpret_cast<const OnigUChar*>(pattern.data());
const auto* end = begin + pattern.size();
if (!regex_valid_utf8(pattern)) {
return regex_error(error, "regex pattern is not valid UTF-8");
}
auto next = std::make_unique<Impl>();
OnigErrorInfo info{};
static std::mutex compile_mutex;
std::lock_guard<std::mutex> lock(compile_mutex);
const int result = onig_new(&next->regex, begin, end, ONIG_OPTION_NONE,
ONIG_ENCODING_UTF8, ONIG_SYNTAX_ONIGURUMA, &info);
if (result != ONIG_NORMAL) {
return regex_onig_error(error, result, &info);
}
impl_ = std::move(next);
return true;
}
bool Regex::find_matches(const std::string& text, std::vector<Match>& matches, std::string* error) const {
matches.clear();
if (error) {
error->clear();
}
if (!impl_) {
return regex_error(error, "regex has not been compiled");
}
if (text.size() > static_cast<size_t>(std::numeric_limits<int>::max())) {
return regex_error(error, "regex input exceeds Oniguruma's offset range");
}
const auto* begin = reinterpret_cast<const OnigUChar*>(text.data());
const auto* end = begin + text.size();
if (!regex_valid_utf8(text)) {
return regex_error(error, "regex input is not valid UTF-8");
}
std::unique_ptr<OnigRegion, RegexRegionDeleter> region(onig_region_new());
if (!region) {
return regex_error(error, "failed to allocate regex match region");
}
size_t position = 0;
while (position <= text.size()) {
const int result = onig_search(impl_->regex, begin, end, begin + position, end,
region.get(), ONIG_OPTION_NONE);
if (result == ONIG_MISMATCH) {
break;
}
if (result < 0) {
matches.clear();
return regex_onig_error(error, result);
}
const size_t match_begin = static_cast<size_t>(region->beg[0]);
const size_t match_end = static_cast<size_t>(region->end[0]);
// Match rust-onig's find_iter: suppress an empty match at the previous match's end.
if (match_begin == match_end && !matches.empty() && matches.back().second == match_end) {
if (position == text.size()) {
break;
}
position += static_cast<size_t>(ONIGENC_MBC_ENC_LEN(ONIG_ENCODING_UTF8, begin + position));
continue;
}
matches.emplace_back(match_begin, match_end);
position = match_end;
}
return true;
}
} // namespace sd
+32
View File
@@ -0,0 +1,32 @@
#ifndef __SD_CORE_REGEX_H__
#define __SD_CORE_REGEX_H__
#include <cstddef>
#include <memory>
#include <string>
#include <utility>
#include <vector>
namespace sd {
class Regex {
struct Impl;
std::unique_ptr<Impl> impl_;
public:
using Match = std::pair<size_t, size_t>;
Regex();
~Regex();
Regex(Regex&&) noexcept;
Regex& operator=(Regex&&) noexcept;
// Failed compilation leaves the previous expression intact.
bool compile(const std::string& pattern, std::string* error = nullptr);
// Matches are non-overlapping UTF-8 byte ranges; each call owns its search state.
bool find_matches(const std::string& text, std::vector<Match>& matches, std::string* error = nullptr) const;
};
} // namespace sd
#endif // __SD_CORE_REGEX_H__
+7
View File
@@ -1,6 +1,8 @@
#ifndef __SD_CORE_RNG_HPP__
#define __SD_CORE_RNG_HPP__
#include <cstdint>
#include <memory>
#include <random>
#include <vector>
@@ -8,6 +10,7 @@ class RNG {
public:
virtual void manual_seed(uint64_t seed) = 0;
virtual std::vector<float> randn(uint32_t n) = 0;
virtual std::shared_ptr<RNG> clone() const = 0;
};
class STDDefaultRNG : public RNG {
@@ -15,6 +18,10 @@ private:
std::default_random_engine generator;
public:
std::shared_ptr<RNG> clone() const override {
return std::make_shared<STDDefaultRNG>(*this);
}
void manual_seed(uint64_t seed) override {
generator.seed((unsigned int)seed);
}
+6
View File
@@ -1,7 +1,9 @@
#ifndef __SD_CORE_RNG_MT19937_HPP__
#define __SD_CORE_RNG_MT19937_HPP__
#include <array>
#include <cmath>
#include <limits>
#include <vector>
#include "core/rng.hpp"
@@ -123,6 +125,10 @@ class MT19937RNG : public RNG {
public:
MT19937RNG(uint64_t seed = 0) { manual_seed(seed); }
std::shared_ptr<RNG> clone() const override {
return std::make_shared<MT19937RNG>(*this);
}
void manual_seed(uint64_t seed) override {
s.seed_ = seed;
s.seeded_ = true;
+31 -60
View File
@@ -1,6 +1,7 @@
#ifndef __SD_CORE_RNG_PHILOX_HPP__
#define __SD_CORE_RNG_PHILOX_HPP__
#include <array>
#include <cmath>
#include <vector>
@@ -14,67 +15,35 @@ private:
uint32_t offset;
private:
std::vector<uint32_t> philox_m = {0xD2511F53, 0xCD9E8D57};
std::vector<uint32_t> philox_w = {0x9E3779B9, 0xBB67AE85};
float two_pow32_inv = 2.3283064e-10f;
float two_pow32_inv_2pi = 2.3283064e-10f * 6.2831855f;
using Counter = std::array<std::vector<uint32_t>, 4>;
std::vector<uint32_t> uint32(uint64_t x) {
std::vector<uint32_t> result(2);
result[0] = static_cast<uint32_t>(x & 0xFFFFFFFF);
result[1] = static_cast<uint32_t>(x >> 32);
return result;
}
std::vector<std::vector<uint32_t>> uint32(const std::vector<uint64_t>& x) {
uint32_t N = (uint32_t)x.size();
std::vector<std::vector<uint32_t>> result(2, std::vector<uint32_t>(N));
for (uint32_t i = 0; i < N; ++i) {
result[0][i] = static_cast<uint32_t>(x[i] & 0xFFFFFFFF);
result[1][i] = static_cast<uint32_t>(x[i] >> 32);
}
return result;
}
static constexpr uint32_t philox_m[2] = {0xD2511F53, 0xCD9E8D57};
static constexpr uint32_t philox_w[2] = {0x9E3779B9, 0xBB67AE85};
float two_pow32_inv = 2.3283064e-10f;
float two_pow32_inv_2pi = 2.3283064e-10f * 6.2831855f;
// A single round of the Philox 4x32 random number generator.
void philox4_round(std::vector<std::vector<uint32_t>>& counter,
const std::vector<std::vector<uint32_t>>& key) {
void philox4_round(Counter& counter, uint32_t key0, uint32_t key1) {
uint32_t N = (uint32_t)counter[0].size();
for (uint32_t i = 0; i < N; i++) {
std::vector<uint32_t> v1 = uint32(static_cast<uint64_t>(counter[0][i]) * static_cast<uint64_t>(philox_m[0]));
std::vector<uint32_t> v2 = uint32(static_cast<uint64_t>(counter[2][i]) * static_cast<uint64_t>(philox_m[1]));
const uint64_t v1 = static_cast<uint64_t>(counter[0][i]) * static_cast<uint64_t>(philox_m[0]);
const uint64_t v2 = static_cast<uint64_t>(counter[2][i]) * static_cast<uint64_t>(philox_m[1]);
counter[0][i] = v2[1] ^ counter[1][i] ^ key[0][i];
counter[1][i] = v2[0];
counter[2][i] = v1[1] ^ counter[3][i] ^ key[1][i];
counter[3][i] = v1[0];
counter[0][i] = static_cast<uint32_t>(v2 >> 32) ^ counter[1][i] ^ key0;
counter[1][i] = static_cast<uint32_t>(v2);
counter[2][i] = static_cast<uint32_t>(v1 >> 32) ^ counter[3][i] ^ key1;
counter[3][i] = static_cast<uint32_t>(v1);
}
}
// Generates 32-bit random numbers using the Philox 4x32 random number generator.
// Parameters:
// counter : A 4xN array of 32-bit integers representing the counter values (offset into generation).
// key : A 2xN array of 32-bit integers representing the key values (seed).
// rounds : The number of rounds to perform.
// Returns:
// std::vector<std::vector<uint32_t>>: A 4xN array of 32-bit integers containing the generated random numbers.
std::vector<std::vector<uint32_t>> philox4_32(std::vector<std::vector<uint32_t>>& counter,
std::vector<std::vector<uint32_t>>& key,
int rounds = 10) {
uint32_t N = (uint32_t)counter[0].size();
void philox4_32(Counter& counter, uint32_t key0, uint32_t key1, int rounds = 10) {
for (int i = 0; i < rounds - 1; ++i) {
philox4_round(counter, key);
for (uint32_t j = 0; j < N; ++j) {
key[0][j] += philox_w[0];
key[1][j] += philox_w[1];
}
philox4_round(counter, key0, key1);
key0 += philox_w[0];
key1 += philox_w[1];
}
philox4_round(counter, key);
return counter;
philox4_round(counter, key0, key1);
}
float box_muller(float x, float y) {
@@ -93,33 +62,35 @@ public:
this->offset = 0;
}
std::shared_ptr<RNG> clone() const override {
return std::make_shared<PhiloxRNG>(*this);
}
void manual_seed(uint64_t seed) override {
this->seed = seed;
this->offset = 0;
}
std::vector<float> randn(uint32_t n) override {
std::vector<std::vector<uint32_t>> counter(4, std::vector<uint32_t>(n, 0));
for (uint32_t i = 0; i < n; i++) {
counter[0][i] = this->offset;
}
Counter counter;
counter[0].resize(n, this->offset);
counter[1].resize(n);
counter[2].resize(n);
counter[3].resize(n);
for (uint32_t i = 0; i < n; i++) {
counter[2][i] = i;
}
this->offset += 1;
std::vector<uint64_t> key(n, this->seed);
std::vector<std::vector<uint32_t>> key_uint32 = uint32(key);
philox4_32(counter, static_cast<uint32_t>(this->seed), static_cast<uint32_t>(this->seed >> 32));
std::vector<std::vector<uint32_t>> g = philox4_32(counter, key_uint32);
std::vector<float> result;
std::vector<float> result(n);
for (uint32_t i = 0; i < n; ++i) {
result.push_back(box_muller((float)g[0][i], (float)g[1][i]));
result[i] = box_muller((float)counter[0][i], (float)counter[1][i]);
}
return result;
}
};
#endif // __SD_CORE_RNG_PHILOX_HPP__
#endif // __SD_CORE_RNG_PHILOX_HPP__
+121 -89
View File
@@ -16,6 +16,7 @@
#include <utility>
#include <vector>
#include "core/parallel.h"
#include "core/rng.hpp"
namespace sd {
@@ -59,6 +60,15 @@ namespace sd {
return numel;
}
template <typename F>
inline void tensor_for_each(int64_t count, F&& fn, int64_t grain_size = 65536) {
parallel_for(0, count, grain_size, [&](int64_t begin, int64_t end) {
for (int64_t i = begin; i < end; ++i) {
fn(i);
}
});
}
template <typename T>
class Tensor {
public:
@@ -230,7 +240,10 @@ namespace sd {
}
void fill_(const T& value) {
std::fill(data_.begin(), data_.end(), value);
const T fill_value = value;
parallel_for(0, numel(), 65536, [&](int64_t begin, int64_t end) {
std::fill_n(data_.data() + begin, end - begin, fill_value);
});
}
Tensor& masked_fill_(const Tensor<uint8_t>& mask, const T& value);
@@ -390,7 +403,7 @@ namespace sd {
tensor_shape_to_string(lhs) + ", rhs_shape=" +
tensor_shape_to_string(rhs));
}
shape[i] = std::max(lhs_dim, rhs_dim);
shape[i] = lhs_dim == 1 ? rhs_dim : lhs_dim;
}
return shape;
}
@@ -425,39 +438,55 @@ namespace sd {
const std::vector<int64_t>& rhs_shape_raw,
const std::vector<int64_t>& rhs_strides_raw,
F&& fn) {
const size_t ndim = out_shape.size();
std::vector<int64_t> out_strides = tensor_compute_strides(out_shape);
std::vector<int64_t> lhs_shape(ndim, 1);
std::vector<int64_t> lhs_strides(ndim, 0);
std::vector<int64_t> rhs_shape(ndim, 1);
std::vector<int64_t> rhs_strides(ndim, 0);
for (size_t i = 0; i < lhs_shape_raw.size(); ++i) {
lhs_shape[i] = lhs_shape_raw[i];
lhs_strides[i] = lhs_strides_raw[i];
}
for (size_t i = 0; i < rhs_shape_raw.size(); ++i) {
rhs_shape[i] = rhs_shape_raw[i];
rhs_strides[i] = rhs_strides_raw[i];
}
const int64_t numel = tensor_numel(out_shape);
for (int64_t flat = 0; flat < numel; ++flat) {
int64_t remaining = flat;
int64_t lhs_offset = 0;
int64_t rhs_offset = 0;
for (size_t i = ndim; i-- > 0;) {
int64_t coord = remaining / out_strides[i];
remaining %= out_strides[i];
if (lhs_shape[i] != 1) {
lhs_offset += coord * lhs_strides[i];
const int64_t numel = tensor_numel(out_shape);
const size_t ndim = out_shape.size();
auto broadcast_strides = [&](const std::vector<int64_t>& shape,
const std::vector<int64_t>& strides) {
if ((numel != 0 && tensor_numel(shape) == 0) || strides.size() != shape.size()) {
tensor_throw_invalid_argument("Tensor broadcast requires non-empty inputs and matching strides");
}
std::vector<int64_t> result(ndim, 0);
for (size_t i = 0; i < std::max(ndim, shape.size()); ++i) {
const int64_t input_dim = i < shape.size() ? shape[i] : 1;
const int64_t output_dim = i < ndim ? out_shape[i] : 1;
if (input_dim != 1 && input_dim != output_dim) {
tensor_throw_invalid_argument("Tensor broadcast cannot expand the destination: input_shape=" +
tensor_shape_to_string(shape) + ", output_shape=" +
tensor_shape_to_string(out_shape));
}
if (rhs_shape[i] != 1) {
rhs_offset += coord * rhs_strides[i];
if (i < ndim && input_dim != 1) {
result[i] = strides[i];
}
}
fn(flat, lhs_offset, rhs_offset);
return result;
};
const auto lhs_strides = broadcast_strides(lhs_shape_raw, lhs_strides_raw);
const auto rhs_strides = broadcast_strides(rhs_shape_raw, rhs_strides_raw);
if (numel == 0) {
return;
}
parallel_for(0, numel, 16384, [&](int64_t begin, int64_t end) {
auto coord = tensor_unravel_index(begin, out_shape);
int64_t lhs_offset = 0;
int64_t rhs_offset = 0;
for (size_t i = 0; i < ndim; ++i) {
lhs_offset += coord[i] * lhs_strides[i];
rhs_offset += coord[i] * rhs_strides[i];
}
for (int64_t flat = begin; flat < end; ++flat) {
fn(flat, lhs_offset, rhs_offset);
for (size_t i = 0; i < ndim; ++i) {
lhs_offset += lhs_strides[i];
rhs_offset += rhs_strides[i];
if (++coord[i] < out_shape[i]) {
break;
}
coord[i] = 0;
lhs_offset -= out_shape[i] * lhs_strides[i];
rhs_offset -= out_shape[i] * rhs_strides[i];
}
}
});
}
template <typename T>
@@ -469,6 +498,7 @@ namespace sd {
const std::vector<int64_t> data_strides = tensor_compute_strides(shape_);
const std::vector<int64_t> mask_strides = tensor_compute_strides(mask.shape());
const uint8_t* mask_data = mask.data();
const T fill_value = value;
tensor_for_each_broadcast_offset(shape_,
shape_,
data_strides,
@@ -476,7 +506,7 @@ namespace sd {
mask_strides,
[&](int64_t, int64_t data_offset, int64_t mask_offset) {
if (mask_data[mask_offset] != 0) {
data_[static_cast<size_t>(data_offset)] = value;
data_[static_cast<size_t>(data_offset)] = fill_value;
}
});
return *this;
@@ -486,9 +516,9 @@ namespace sd {
inline Tensor<uint8_t> operator<(const Tensor<T>& lhs, Scalar rhs) {
Tensor<uint8_t> result(lhs.shape());
const T value = static_cast<T>(rhs);
for (int64_t i = 0; i < lhs.numel(); ++i) {
result[i] = lhs[i] < value ? 1 : 0;
}
tensor_for_each(lhs.numel(), [&](int64_t i) {
result.data()[i] = lhs.data()[i] < value ? 1 : 0;
});
return result;
}
@@ -496,9 +526,9 @@ namespace sd {
inline Tensor<uint8_t> operator<(Scalar lhs, const Tensor<T>& rhs) {
Tensor<uint8_t> result(rhs.shape());
const T value = static_cast<T>(lhs);
for (int64_t i = 0; i < rhs.numel(); ++i) {
result[i] = value < rhs[i] ? 1 : 0;
}
tensor_for_each(rhs.numel(), [&](int64_t i) {
result.data()[i] = value < rhs.data()[i] ? 1 : 0;
});
return result;
}
@@ -516,7 +546,7 @@ namespace sd {
rhs.shape(),
rhs_strides,
[&](int64_t flat, int64_t lhs_offset, int64_t rhs_offset) {
result[flat] = lhs_data[lhs_offset] < rhs_data[rhs_offset] ? 1 : 0;
result.data()[flat] = lhs_data[lhs_offset] < rhs_data[rhs_offset] ? 1 : 0;
});
return result;
}
@@ -524,9 +554,9 @@ namespace sd {
template <typename T>
inline Tensor<T>& operator+=(Tensor<T>& lhs, const Tensor<T>& rhs) {
if (lhs.shape() == rhs.shape()) {
for (int64_t i = 0; i < lhs.numel(); ++i) {
lhs[i] += rhs[i];
}
tensor_for_each(lhs.numel(), [&](int64_t i) {
lhs.data()[i] += rhs.data()[i];
});
return lhs;
}
tensor_broadcast_shape(lhs.shape(), rhs.shape());
@@ -539,7 +569,7 @@ namespace sd {
rhs.shape(),
rhs_strides,
[&](int64_t, int64_t lhs_offset, int64_t rhs_offset) {
lhs[static_cast<int64_t>(lhs_offset)] += rhs_data[rhs_offset];
lhs.data()[lhs_offset] += rhs_data[rhs_offset];
});
return lhs;
}
@@ -547,18 +577,18 @@ namespace sd {
template <typename T, typename Scalar, typename = std::enable_if_t<std::is_arithmetic<Scalar>::value>>
inline Tensor<T>& operator+=(Tensor<T>& lhs, Scalar rhs) {
const T value = static_cast<T>(rhs);
for (int64_t i = 0; i < lhs.numel(); ++i) {
lhs[i] += value;
}
tensor_for_each(lhs.numel(), [&](int64_t i) {
lhs.data()[i] += value;
});
return lhs;
}
template <typename T>
inline Tensor<T>& operator-=(Tensor<T>& lhs, const Tensor<T>& rhs) {
if (lhs.shape() == rhs.shape()) {
for (int64_t i = 0; i < lhs.numel(); ++i) {
lhs[i] -= rhs[i];
}
tensor_for_each(lhs.numel(), [&](int64_t i) {
lhs.data()[i] -= rhs.data()[i];
});
return lhs;
}
tensor_broadcast_shape(lhs.shape(), rhs.shape());
@@ -571,7 +601,7 @@ namespace sd {
rhs.shape(),
rhs_strides,
[&](int64_t, int64_t lhs_offset, int64_t rhs_offset) {
lhs[static_cast<int64_t>(lhs_offset)] -= rhs_data[rhs_offset];
lhs.data()[lhs_offset] -= rhs_data[rhs_offset];
});
return lhs;
}
@@ -579,18 +609,18 @@ namespace sd {
template <typename T, typename Scalar, typename = std::enable_if_t<std::is_arithmetic<Scalar>::value>>
inline Tensor<T>& operator-=(Tensor<T>& lhs, Scalar rhs) {
const T value = static_cast<T>(rhs);
for (int64_t i = 0; i < lhs.numel(); ++i) {
lhs[i] -= value;
}
tensor_for_each(lhs.numel(), [&](int64_t i) {
lhs.data()[i] -= value;
});
return lhs;
}
template <typename T>
inline Tensor<T>& operator*=(Tensor<T>& lhs, const Tensor<T>& rhs) {
if (lhs.shape() == rhs.shape()) {
for (int64_t i = 0; i < lhs.numel(); ++i) {
lhs[i] *= rhs[i];
}
tensor_for_each(lhs.numel(), [&](int64_t i) {
lhs.data()[i] *= rhs.data()[i];
});
return lhs;
}
tensor_broadcast_shape(lhs.shape(), rhs.shape());
@@ -603,7 +633,7 @@ namespace sd {
rhs.shape(),
rhs_strides,
[&](int64_t, int64_t lhs_offset, int64_t rhs_offset) {
lhs[static_cast<int64_t>(lhs_offset)] *= rhs_data[rhs_offset];
lhs.data()[lhs_offset] *= rhs_data[rhs_offset];
});
return lhs;
}
@@ -611,18 +641,18 @@ namespace sd {
template <typename T, typename Scalar, typename = std::enable_if_t<std::is_arithmetic<Scalar>::value>>
inline Tensor<T>& operator*=(Tensor<T>& lhs, Scalar rhs) {
const T value = static_cast<T>(rhs);
for (int64_t i = 0; i < lhs.numel(); ++i) {
lhs[i] *= value;
}
tensor_for_each(lhs.numel(), [&](int64_t i) {
lhs.data()[i] *= value;
});
return lhs;
}
template <typename T>
inline Tensor<T>& operator/=(Tensor<T>& lhs, const Tensor<T>& rhs) {
if (lhs.shape() == rhs.shape()) {
for (int64_t i = 0; i < lhs.numel(); ++i) {
lhs[i] /= rhs[i];
}
tensor_for_each(lhs.numel(), [&](int64_t i) {
lhs.data()[i] /= rhs.data()[i];
});
return lhs;
}
tensor_broadcast_shape(lhs.shape(), rhs.shape());
@@ -635,7 +665,7 @@ namespace sd {
rhs.shape(),
rhs_strides,
[&](int64_t, int64_t lhs_offset, int64_t rhs_offset) {
lhs[static_cast<int64_t>(lhs_offset)] /= rhs_data[rhs_offset];
lhs.data()[lhs_offset] /= rhs_data[rhs_offset];
});
return lhs;
}
@@ -643,9 +673,9 @@ namespace sd {
template <typename T, typename Scalar, typename = std::enable_if_t<std::is_arithmetic<Scalar>::value>>
inline Tensor<T>& operator/=(Tensor<T>& lhs, Scalar rhs) {
const T value = static_cast<T>(rhs);
for (int64_t i = 0; i < lhs.numel(); ++i) {
lhs[i] /= value;
}
tensor_for_each(lhs.numel(), [&](int64_t i) {
lhs.data()[i] /= value;
});
return lhs;
}
@@ -664,7 +694,7 @@ namespace sd {
rhs.shape(),
rhs_strides,
[&](int64_t flat, int64_t lhs_offset, int64_t rhs_offset) {
result[flat] = lhs_data[lhs_offset] + rhs_data[rhs_offset];
result.data()[flat] = lhs_data[lhs_offset] + rhs_data[rhs_offset];
});
return result;
}
@@ -699,7 +729,7 @@ namespace sd {
rhs.shape(),
rhs_strides,
[&](int64_t flat, int64_t lhs_offset, int64_t rhs_offset) {
result[flat] = lhs_data[lhs_offset] - rhs_data[rhs_offset];
result.data()[flat] = lhs_data[lhs_offset] - rhs_data[rhs_offset];
});
return result;
}
@@ -717,9 +747,9 @@ namespace sd {
inline Tensor<T> operator-(Scalar lhs, const Tensor<T>& rhs) {
Tensor<T> result = rhs;
const T value = static_cast<T>(lhs);
for (int64_t i = 0; i < result.numel(); ++i) {
result[i] = value - result[i];
}
tensor_for_each(result.numel(), [&](int64_t i) {
result.data()[i] = value - result.data()[i];
});
return result;
}
@@ -738,7 +768,7 @@ namespace sd {
rhs.shape(),
rhs_strides,
[&](int64_t flat, int64_t lhs_offset, int64_t rhs_offset) {
result[flat] = lhs_data[lhs_offset] * rhs_data[rhs_offset];
result.data()[flat] = lhs_data[lhs_offset] * rhs_data[rhs_offset];
});
return result;
}
@@ -773,7 +803,7 @@ namespace sd {
rhs.shape(),
rhs_strides,
[&](int64_t flat, int64_t lhs_offset, int64_t rhs_offset) {
result[flat] = lhs_data[lhs_offset] / rhs_data[rhs_offset];
result.data()[flat] = lhs_data[lhs_offset] / rhs_data[rhs_offset];
});
return result;
}
@@ -791,18 +821,18 @@ namespace sd {
inline Tensor<T> operator/(Scalar lhs, const Tensor<T>& rhs) {
Tensor<T> result = rhs;
const T value = static_cast<T>(lhs);
for (int64_t i = 0; i < result.numel(); ++i) {
result[i] = value / result[i];
}
tensor_for_each(result.numel(), [&](int64_t i) {
result.data()[i] = value / result.data()[i];
});
return result;
}
template <typename T>
inline Tensor<T> operator-(const Tensor<T>& tensor) {
Tensor<T> result = tensor;
for (int64_t i = 0; i < result.numel(); ++i) {
result[i] = -result[i];
}
tensor_for_each(result.numel(), [&](int64_t i) {
result.data()[i] = -result.data()[i];
});
return result;
}
@@ -1067,9 +1097,11 @@ namespace sd {
template <typename T>
inline Tensor<T> exp(const Tensor<T>& input) {
Tensor<T> output(input.shape());
for (int64_t i = 0; i < input.numel(); ++i) {
output[i] = static_cast<T>(std::exp(static_cast<double>(input[i])));
}
tensor_for_each(
input.numel(), [&](int64_t i) {
output.data()[i] = static_cast<T>(std::exp(static_cast<double>(input.data()[i])));
},
4096);
return output;
}
@@ -1079,18 +1111,18 @@ namespace sd {
tensor_throw_invalid_argument("Tensor clamp requires min_value <= max_value");
}
Tensor<T> output(input.shape());
for (int64_t i = 0; i < input.numel(); ++i) {
output[i] = std::clamp(input[i], min_value, max_value);
}
tensor_for_each(input.numel(), [&](int64_t i) {
output.data()[i] = std::clamp(input.data()[i], min_value, max_value);
});
return output;
}
template <typename T>
inline Tensor<T> round(const Tensor<T>& input) {
Tensor<T> output(input.shape());
for (int64_t i = 0; i < input.numel(); ++i) {
output[i] = static_cast<T>(std::round(static_cast<double>(input[i])));
}
tensor_for_each(input.numel(), [&](int64_t i) {
output.data()[i] = static_cast<T>(std::round(static_cast<double>(input.data()[i])));
});
return output;
}
+82 -9
View File
@@ -414,14 +414,43 @@ std::vector<std::string> split_string(const std::string& str, char delimiter) {
}
ggml_type sd_type_to_ggml_type(sd_type_t sdtype) {
if (sdtype == SD_TYPE_F8_E4M3 || sdtype == SD_TYPE_F8_E5M2) {
#ifndef SD_USE_UPSTREAM_GGML
return sdtype == SD_TYPE_F8_E4M3 ? GGML_TYPE_F8_E4M3 : GGML_TYPE_F8_E5M2;
#else
return GGML_TYPE_COUNT;
#endif
}
const int type_value = static_cast<int>(sdtype);
if (type_value < std::min<int>(SD_TYPE_COUNT, GGML_TYPE_COUNT)) {
if (type_value >= 0 && type_value < std::min<int>(SD_TYPE_COUNT, GGML_TYPE_COUNT)) {
return static_cast<ggml_type>(type_value);
} else {
return GGML_TYPE_COUNT;
}
}
bool validate_tensor_types(sd_type_t type, const char* tensor_type_rules) {
if (type != SD_TYPE_COUNT && sd_type_to_ggml_type(type) == GGML_TYPE_COUNT) {
LOG_ERROR("weight type %s is not supported by this ggml build", sd_type_name(type));
return false;
}
#ifdef SD_USE_UPSTREAM_GGML
for (const auto& rule : split_string(SAFE_STR(tensor_type_rules), ',')) {
const auto pos = rule.find('=');
if (pos != std::string::npos) {
const auto name = rule.substr(pos + 1);
if (name == "f8_e4m3" || name == "f8_e5m2") {
LOG_ERROR("FP8 is not supported by this ggml build (tensor type rule '%s')", rule.c_str());
return false;
}
}
}
#else
GGML_UNUSED(tensor_type_rules);
#endif
return true;
}
KeyValueArgs parse_key_value_args(const char* args, const char* context) {
KeyValueArgs pairs;
@@ -755,6 +784,13 @@ sd::Tensor<float> clip_preprocess(const sd::Tensor<float>& image, int target_wid
int64_t resized_width = static_cast<int64_t>(scale * static_cast<float>(image.shape()[0]));
int64_t resized_height = static_cast<int64_t>(scale * static_cast<float>(image.shape()[1]));
// The resized image must cover the crop window. Floating-point rounding can
// leave a side one pixel short of the crop target (e.g. 730 -> 735.999...
// -> 735 after truncation), so clamp to keep the center crop in bounds.
// Truncation is otherwise preserved to avoid changing existing results.
resized_width = std::max<int64_t>(resized_width, target_width);
resized_height = std::max<int64_t>(resized_height, target_height);
sd::Tensor<float> resized = sd::ops::interpolate(
image,
{resized_width, resized_height, image.shape()[2], image.shape()[3]});
@@ -821,7 +857,11 @@ std::vector<std::pair<std::string, float>> parse_prompt_attention(const std::str
float round_bracket_multiplier = 1.1f;
float square_bracket_multiplier = 1 / 1.1f;
std::regex re_attention(R"(\\\(|\\\)|\\\[|\\\]|\\\\|\\|\(|\[|:([+-]?[.\d]+)\)|\)|\]|\bBREAK\b|[^\\()\[\]:B]+|:|\bB)");
// libstdc++ std::regex recurses per matched character, so unbounded runs
// overflow the stack. Split runs are merged back below.
const int max_plain_text_run = 1024;
std::regex re_attention(R"(\\\(|\\\)|\\\[|\\\]|\\\\|\\|\(|\[|\)|\]|\bBREAK\b|[^\\()\[\]:B]{1,)" +
std::to_string(max_plain_text_run) + R"(}|:|\bB)");
std::regex re_break(R"(\s*\bBREAK\b\s*)");
auto multiply_range = [&](int start_position, float multiplier) {
@@ -830,22 +870,55 @@ std::vector<std::pair<std::string, float>> parse_prompt_attention(const std::str
}
};
// Kept out of the regex: bounding the repetition rejects valid long weights,
// leaving it unbounded overflows the stack.
auto lex_weight = [](const std::string& s, float& value) -> size_t {
size_t end = 0;
if (end < s.size() && (s[end] == '+' || s[end] == '-')) {
++end;
}
while (end < s.size() && (std::isdigit((unsigned char)s[end]) || s[end] == '.')) {
++end;
}
if (end >= s.size() || s[end] != ')') {
return 0;
}
std::string number = s.substr(0, end);
char* number_end = nullptr;
float parsed = std::strtof(number.c_str(), &number_end);
const char* expected = number.c_str() + number.size();
// Without this ".", "+." and "1.2.3" would silently become weights.
if (number.empty() || number_end != expected || !std::isfinite(parsed)) {
return 0;
}
value = parsed;
return end + 1;
};
std::smatch m, m2;
std::string remaining_text = text;
while (std::regex_search(remaining_text, m, re_attention)) {
std::string text = m[0];
std::string weight = m[1];
std::string suffix = m.suffix();
if (text == ":") {
float weight_value = 1.0f;
size_t weight_length = lex_weight(suffix, weight_value);
if (weight_length > 0) {
if (!round_brackets.empty()) {
multiply_range(round_brackets.back(), weight_value);
round_brackets.pop_back();
}
remaining_text = suffix.substr(weight_length);
continue;
}
}
if (text == "(") {
round_brackets.push_back((int)res.size());
} else if (text == "[") {
square_brackets.push_back((int)res.size());
} else if (!weight.empty()) {
if (!round_brackets.empty()) {
multiply_range(round_brackets.back(), std::stof(weight));
round_brackets.pop_back();
}
} else if (text == ")" && !round_brackets.empty()) {
multiply_range(round_brackets.back(), round_bracket_multiplier);
round_brackets.pop_back();
@@ -860,7 +933,7 @@ std::vector<std::pair<std::string, float>> parse_prompt_attention(const std::str
res.push_back({text, 1.0f});
}
remaining_text = m.suffix();
remaining_text = suffix;
}
for (int pos : round_brackets) {
+1
View File
@@ -90,6 +90,7 @@ void log_printf(sd_log_level_t level, const char* file, int line, const char* fo
void sd_ggml_log_callback(ggml_log_level level, const char* text, void*);
ggml_type sd_type_to_ggml_type(sd_type_t sdtype);
bool validate_tensor_types(sd_type_t type, const char* tensor_type_rules);
std::string trim(const std::string& s);
+65 -30
View File
@@ -18,12 +18,15 @@ tokenize_photomaker_trigger(FrozenCLIPEmbedderWithCustomWords& clip_conditioner,
auto tokens_and_weights = clip_conditioner.tokenize(text);
std::vector<int> source_tokens = std::move(tokens_and_weights.first);
std::vector<float> source_weights = std::move(tokens_and_weights.second);
if (source_tokens.empty()) {
return {};
}
if (!source_tokens.empty() && source_tokens.front() == clip_conditioner.tokenizer.BOS_TOKEN_ID) {
if (!source_tokens.empty() && source_tokens.front() == clip_conditioner.tokenizer->BOS_TOKEN_ID) {
source_tokens.erase(source_tokens.begin());
source_weights.erase(source_weights.begin());
}
if (!source_tokens.empty() && source_tokens.back() == clip_conditioner.tokenizer.EOS_TOKEN_ID) {
if (!source_tokens.empty() && source_tokens.back() == clip_conditioner.tokenizer->EOS_TOKEN_ID) {
source_tokens.pop_back();
source_weights.pop_back();
}
@@ -49,12 +52,12 @@ tokenize_photomaker_trigger(FrozenCLIPEmbedderWithCustomWords& clip_conditioner,
weights.push_back(source_weights[i]);
}
clip_conditioner.tokenizer.pad_tokens(tokens,
&weights,
nullptr,
clip_conditioner.text_model->model.n_token,
clip_conditioner.text_model->model.n_token,
true);
clip_conditioner.tokenizer->pad_tokens(tokens,
&weights,
nullptr,
clip_conditioner.text_model->model.n_token,
clip_conditioner.text_model->model.n_token,
true);
std::vector<bool> class_token_mask;
for (int i = 0; i < tokens.size(); i++) {
class_token_mask.push_back(class_idx >= 0 && class_idx + 1 <= i && i < class_idx + 1 + trigger_token_count);
@@ -69,8 +72,14 @@ get_photomaker_condition_with_trigger(FrozenCLIPEmbedderWithCustomWords& clip_co
const ConditionerParams& conditioner_params,
const std::string& trigger_word,
int trigger_token_count) {
auto image_tokens = clip_conditioner.convert_token_to_id(trigger_word);
GGML_ASSERT(image_tokens.size() == 1);
std::vector<int> image_tokens;
if (!clip_conditioner.convert_token_to_id(trigger_word, image_tokens)) {
return {};
}
if (image_tokens.size() != 1) {
LOG_ERROR("PhotoMaker trigger word must encode to one token");
return {};
}
auto tokens_and_weights = tokenize_photomaker_trigger(clip_conditioner,
conditioner_params.text,
trigger_token_count,
@@ -78,27 +87,43 @@ get_photomaker_condition_with_trigger(FrozenCLIPEmbedderWithCustomWords& clip_co
std::vector<int>& tokens = std::get<0>(tokens_and_weights);
std::vector<float>& weights = std::get<1>(tokens_and_weights);
std::vector<bool>& trigger_mask = std::get<2>(tokens_and_weights);
auto cond = clip_conditioner.get_learned_condition_common(n_threads,
tokens,
weights,
conditioner_params.clip_skip,
conditioner_params.width,
conditioner_params.height,
conditioner_params.zero_out_masked);
if (tokens.empty()) {
return {};
}
auto cond = clip_conditioner.get_learned_condition_common(n_threads,
tokens,
weights,
conditioner_params.clip_skip,
conditioner_params.width,
conditioner_params.height,
conditioner_params.zero_out_masked);
return std::make_tuple(std::move(cond), trigger_mask);
}
static std::string remove_photomaker_trigger_from_prompt(FrozenCLIPEmbedderWithCustomWords& clip_conditioner,
const std::string& prompt,
const std::string& trigger_word) {
auto image_tokens = clip_conditioner.convert_token_to_id(trigger_word);
GGML_ASSERT(image_tokens.size() == 1);
static bool remove_photomaker_trigger_from_prompt(FrozenCLIPEmbedderWithCustomWords& clip_conditioner,
const std::string& prompt,
const std::string& trigger_word,
std::string& result) {
std::vector<int> image_tokens;
if (!clip_conditioner.convert_token_to_id(trigger_word, image_tokens)) {
return false;
}
if (image_tokens.size() != 1) {
LOG_ERROR("PhotoMaker trigger word must encode to one token");
return false;
}
auto tokens_and_weights = clip_conditioner.tokenize(prompt);
std::vector<int>& tokens = tokens_and_weights.first;
auto it = std::find(tokens.begin(), tokens.end(), image_tokens[0]);
GGML_ASSERT(it != tokens.end());
if (tokens.empty()) {
return false;
}
auto it = std::find(tokens.begin(), tokens.end(), image_tokens[0]);
if (it == tokens.end()) {
LOG_ERROR("PhotoMaker trigger word was not found in tokenized prompt");
return false;
}
tokens.erase(it);
return clip_conditioner.decode(tokens);
return clip_conditioner.decode(tokens, result);
}
struct PhotoMakerExtension : public GenerationExtension {
@@ -135,6 +160,7 @@ struct PhotoMakerExtension : public GenerationExtension {
pm_version,
20.f,
ctx.model_manager);
pmid_model->set_scale_overrides(ctx.params->linear_scale, ctx.params->attn_scale);
if (pm_version == PM_VERSION_2) {
LOG_INFO("using PhotoMaker Version 2");
}
@@ -222,6 +248,10 @@ struct PhotoMakerExtension : public GenerationExtension {
trigger_token_count);
SDCondition prepared_id_condition = std::get<0>(cond_tup);
auto class_tokens_mask = std::get<1>(cond_tup);
if (prepared_id_condition.empty()) {
LOG_ERROR("failed to encode PhotoMaker prompt");
return false;
}
if (std::find(class_tokens_mask.begin(), class_tokens_mask.end(), true) == class_tokens_mask.end()) {
LOG_WARN("PhotoMaker trigger word '%s' was not found in prompt", trigger_word.c_str());
LOG_WARN("Turn off PhotoMaker for this request");
@@ -262,11 +292,16 @@ struct PhotoMakerExtension : public GenerationExtension {
prepared_id_condition.c_crossattn = std::move(res);
int64_t t1 = ggml_time_ms();
id_condition = std::move(prepared_id_condition);
start_merge_step = int(ctx.pm_params.style_strength / 100.f * ctx.total_steps);
ctx.condition_params.text = remove_photomaker_trigger_from_prompt(*clip_conditioner,
ctx.condition_params.text,
trigger_word);
std::string prompt;
if (!remove_photomaker_trigger_from_prompt(*clip_conditioner,
ctx.condition_params.text,
trigger_word,
prompt)) {
return false;
}
id_condition = std::move(prepared_id_condition);
start_merge_step = int(ctx.pm_params.style_strength / 100.f * ctx.total_steps);
ctx.condition_params.text = std::move(prompt);
LOG_INFO("Photomaker ID Stacking, taking %" PRId64 " ms", t1 - t0);
LOG_INFO("PHOTOMAKER: start_merge_step: %d", start_merge_step);
+11 -3
View File
@@ -35,9 +35,11 @@ enum SDVersion {
VERSION_WAN2,
VERSION_WAN2_2_I2V,
VERSION_WAN2_2_TI2V,
VERSION_WAN2_2_S2V,
VERSION_LINGBOT_VIDEO,
VERSION_QWEN_IMAGE,
VERSION_QWEN_IMAGE_LAYERED,
VERSION_QWEN_IMAGE_2_1,
VERSION_HUNYUAN_VIDEO,
VERSION_ANIMA,
VERSION_FLUX2,
@@ -57,6 +59,7 @@ enum SDVersion {
VERSION_SEFI_IMAGE,
VERSION_KREA2,
VERSION_MAGE_FLOW,
VERSION_SENSENOVA_U1_5,
VERSION_ESRGAN,
VERSION_COUNT,
};
@@ -129,7 +132,7 @@ static inline bool sd_version_is_minimax_h3(SDVersion version) {
}
static inline bool sd_version_is_wan(SDVersion version) {
if (version == VERSION_WAN2 || version == VERSION_WAN2_2_I2V || version == VERSION_WAN2_2_TI2V) {
if (version == VERSION_WAN2 || version == VERSION_WAN2_2_I2V || version == VERSION_WAN2_2_TI2V || version == VERSION_WAN2_2_S2V) {
return true;
}
return false;
@@ -143,7 +146,7 @@ static inline bool sd_version_is_lingbot_video(SDVersion version) {
}
static inline bool sd_version_is_qwen_image(SDVersion version) {
if (version == VERSION_QWEN_IMAGE || version == VERSION_QWEN_IMAGE_LAYERED) {
if (version == VERSION_QWEN_IMAGE || version == VERSION_QWEN_IMAGE_LAYERED || version == VERSION_QWEN_IMAGE_2_1) {
return true;
}
return false;
@@ -237,6 +240,10 @@ static inline bool sd_version_is_mage_flow(SDVersion version) {
return version == VERSION_MAGE_FLOW;
}
static inline bool sd_version_is_sensenova_u1(SDVersion version) {
return version == VERSION_SENSENOVA_U1_5;
}
static inline bool sd_version_uses_flux_vae(SDVersion version) {
if (sd_version_is_flux(version) || sd_version_is_z_image(version) || sd_version_is_boogu_image(version) || sd_version_is_longcat(version)) {
return true;
@@ -295,7 +302,8 @@ static inline bool sd_version_is_dit(SDVersion version) {
sd_version_is_ideogram4(version) ||
sd_version_is_sefi_image(version) ||
sd_version_is_krea2(version) ||
sd_version_is_mage_flow(version)) {
sd_version_is_mage_flow(version) ||
sd_version_is_sensenova_u1(version)) {
return true;
}
return false;
+1 -1
View File
@@ -95,7 +95,7 @@ namespace IPAdapter {
int64_t L = kv->ne[1];
ggml_tensor* k = ggml_cont(ctx->ggml_ctx, ggml_view_3d(ctx->ggml_ctx, kv, dim, L, N, kv->nb[1], kv->nb[2], 0));
ggml_tensor* v = ggml_cont(ctx->ggml_ctx, ggml_view_3d(ctx->ggml_ctx, kv, dim, L, N, kv->nb[1], kv->nb[2], dim * kv->nb[0]));
ggml_tensor* attn = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, q, k, v, heads, nullptr, false, false);
ggml_tensor* attn = ggml_ext_attention_ext(ctx, q, k, v, heads, nullptr, false, false);
attn = to_out->forward(ctx, attn);
latents = ggml_add(ctx->ggml_ctx, latents, attn);
+1 -1
View File
@@ -67,7 +67,7 @@ struct LoraModel : public GGMLRunner {
std::map<std::string, ggml_tensor*> scalars;
std::set<std::string> scalar_names;
for (const auto& [name, source] : sources) {
if (is_unused_tensor(name) || (filter && !filter(name)))
if (filter && !filter(name))
continue;
const bool scalar = source.nelements() == 1 && (ends_with(name, ".alpha") || ends_with(name, ".scale"));
auto* tensor = ggml_new_tensor(params_ctx, scalar ? GGML_TYPE_F32 : source.type, source.n_dims, source.ne);
+5 -6
View File
@@ -63,12 +63,11 @@ public:
k = ggml_cont(ctx->ggml_ctx, k);
v = ggml_cont(ctx->ggml_ctx, v);
ggml_tensor* attn_out = ggml_ext_attention_ext(
ctx->ggml_ctx, ctx->backend,
q, k, v,
heads,
/*mask=*/nullptr,
/*diag_mask_inf=*/false);
ggml_tensor* attn_out = ggml_ext_attention_ext(ctx,
q, k, v,
heads,
/*mask=*/nullptr,
/*diag_mask_inf=*/false);
ggml_tensor* out = to_out->forward(ctx, attn_out);
return out;
+413
View File
@@ -0,0 +1,413 @@
#ifndef __SD_MODEL_AUDIO_WAV2VEC2_HPP__
#define __SD_MODEL_AUDIO_WAV2VEC2_HPP__
#include <algorithm>
#include <cinttypes>
#include <cmath>
#include <cstdio>
#include <map>
#include <memory>
#include <string>
#include <vector>
#include "core/ggml_extend.h"
#include "core/ggml_runner.h"
#include "model.h"
#include "model/common/ggml_block.hpp"
namespace Wav2Vec2 {
struct Wav2Vec2Config {
int64_t embed_dim = 1024;
int64_t conv_dim = 512;
int num_heads = 16;
int num_layers = 24;
std::string feat_extract_norm = "layer";
bool conv_bias = true;
bool do_normalize = true;
bool do_stable_layer_norm = true;
static Wav2Vec2Config detect_from_weights(const String2TensorStorage& tensor_storage_map, const std::string& prefix) {
Wav2Vec2Config config;
auto it = tensor_storage_map.find(prefix + "encoder.layer_norm.bias");
if (it == tensor_storage_map.end()) {
LOG_WARN("wav2vec2: %sencoder.layer_norm.bias not found, using large defaults", prefix.c_str());
return config;
}
config.embed_dim = it->second.ne[0];
if (config.embed_dim == 1024) {
config.embed_dim = 1024;
config.num_heads = 16;
config.num_layers = 24;
config.feat_extract_norm = "layer";
config.conv_bias = true;
config.do_normalize = true;
config.do_stable_layer_norm = true;
} else if (config.embed_dim == 768) {
config.embed_dim = 768;
config.num_heads = 12;
config.num_layers = 12;
config.feat_extract_norm = "group";
config.conv_bias = false;
config.do_normalize = false;
config.do_stable_layer_norm = false;
} else {
LOG_WARN("wav2vec2: unsupported embed_dim %" PRId64 ", using large defaults", config.embed_dim);
config.embed_dim = 1024;
}
return config;
}
};
struct Wav2Vec2NoLayerNormConvLayer : public UnaryBlock {
Wav2Vec2NoLayerNormConvLayer(int64_t in_channels, int64_t out_channels, int kernel_size, int stride, bool bias) {
blocks["conv"] = std::make_shared<Conv1d>(in_channels, out_channels, kernel_size, stride, 0, 1, 1, bias, true);
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) override {
auto conv = std::dynamic_pointer_cast<Conv1d>(blocks["conv"]);
x = conv->forward(ctx, x);
return ggml_gelu_erf_inplace(ctx->ggml_ctx, ggml_ext_cont(ctx->ggml_ctx, x));
}
};
struct Wav2Vec2LayerNormConvLayer : public UnaryBlock {
Wav2Vec2LayerNormConvLayer(int64_t in_channels, int64_t out_channels, int kernel_size, int stride, bool bias) {
blocks["conv"] = std::make_shared<Conv1d>(in_channels, out_channels, kernel_size, stride, 0, 1, 1, bias, true);
blocks["layer_norm"] = std::make_shared<LayerNorm>(out_channels);
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) override {
auto conv = std::dynamic_pointer_cast<Conv1d>(blocks["conv"]);
auto layer_norm = std::dynamic_pointer_cast<LayerNorm>(blocks["layer_norm"]);
x = conv->forward(ctx, x);
// LayerNorm normalizes channels: [N, C, L] -> [N, L, C].
x = ggml_permute(ctx->ggml_ctx, x, 1, 0, 2, 3);
x = layer_norm->forward(ctx, x);
x = ggml_permute(ctx->ggml_ctx, x, 1, 0, 2, 3);
return ggml_gelu_erf_inplace(ctx->ggml_ctx, ggml_ext_cont(ctx->ggml_ctx, x));
}
};
struct Wav2Vec2GroupNormConvLayer : public UnaryBlock {
Wav2Vec2GroupNormConvLayer(int64_t in_channels, int64_t out_channels, int kernel_size, int stride, bool bias) {
blocks["conv"] = std::make_shared<Conv1d>(in_channels, out_channels, kernel_size, stride, 0, 1, 1, bias, true);
blocks["layer_norm"] = std::make_shared<GroupNorm>((int)out_channels, out_channels, 1e-05f);
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) override {
auto conv = std::dynamic_pointer_cast<Conv1d>(blocks["conv"]);
auto layer_norm = std::dynamic_pointer_cast<GroupNorm>(blocks["layer_norm"]);
x = conv->forward(ctx, x);
// ggml GroupNorm needs [N, C, H, W], with H=1 for audio.
x = ggml_reshape_4d(ctx->ggml_ctx, x, x->ne[0], 1, x->ne[1], x->ne[2]);
x = layer_norm->forward(ctx, x);
x = ggml_reshape_3d(ctx->ggml_ctx, x, x->ne[0], x->ne[2], x->ne[3]);
return ggml_gelu_erf_inplace(ctx->ggml_ctx, ggml_ext_cont(ctx->ggml_ctx, x));
}
};
struct Wav2Vec2FeatureEncoder : public UnaryBlock {
Wav2Vec2FeatureEncoder(const Wav2Vec2Config& config) {
GGML_ASSERT(config.feat_extract_norm == "layer" || config.feat_extract_norm == "group");
const int kernels[7] = {10, 3, 3, 3, 3, 2, 2};
const int strides[7] = {5, 2, 2, 2, 2, 2, 2};
int64_t in_channels = 1;
for (int i = 0; i < 7; ++i) {
const std::string name = "conv_layers." + std::to_string(i);
if (config.feat_extract_norm == "layer") {
blocks[name] = std::make_shared<Wav2Vec2LayerNormConvLayer>(in_channels, config.conv_dim, kernels[i], strides[i], config.conv_bias);
} else if (i == 0) {
blocks[name] = std::make_shared<Wav2Vec2GroupNormConvLayer>(in_channels, config.conv_dim, kernels[i], strides[i], config.conv_bias);
} else {
blocks[name] = std::make_shared<Wav2Vec2NoLayerNormConvLayer>(in_channels, config.conv_dim, kernels[i], strides[i], config.conv_bias);
}
in_channels = config.conv_dim;
}
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) override {
for (int i = 0; i < 7; ++i) {
auto conv = std::dynamic_pointer_cast<UnaryBlock>(blocks["conv_layers." + std::to_string(i)]);
x = conv->forward(ctx, x);
}
return ggml_permute(ctx->ggml_ctx, x, 1, 0, 2, 3);
}
};
struct Wav2Vec2FeatureProjection : public UnaryBlock {
Wav2Vec2FeatureProjection(const Wav2Vec2Config& config) {
blocks["layer_norm"] = std::make_shared<LayerNorm>(config.conv_dim);
blocks["projection"] = std::make_shared<Linear>(config.conv_dim, config.embed_dim);
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) {
auto ln = std::dynamic_pointer_cast<LayerNorm>(blocks["layer_norm"]);
auto projection = std::dynamic_pointer_cast<Linear>(blocks["projection"]);
x = ln->forward(ctx, x);
x = projection->forward(ctx, x);
return x;
}
};
class Wav2Vec2PositionalConvEmbedding : public UnaryBlock {
private:
int64_t embed_dim_;
static constexpr int groups_ = 16;
static constexpr int kernel_size_ = 128;
std::string weight_g_name_;
std::string weight_v_name_;
ggml_tensor* weight(GGMLRunnerContext* ctx) {
auto g = params[weight_g_name_];
auto v = ggml_cast(ctx->ggml_ctx, params[weight_v_name_], GGML_TYPE_F32);
auto squared = ggml_mul(ctx->ggml_ctx, v, v);
// PyTorch weight_norm(dim=2) reduces both channel axes, retaining each kernel tap.
squared = ggml_cont(ctx->ggml_ctx, ggml_permute(ctx->ggml_ctx, squared, 2, 0, 1, 3));
squared = ggml_reshape_2d(ctx->ggml_ctx, squared, embed_dim_ / groups_ * embed_dim_, kernel_size_);
auto norm = ggml_sqrt(ctx->ggml_ctx, ggml_sum_rows(ctx->ggml_ctx, squared));
norm = ggml_reshape_3d(ctx->ggml_ctx, norm, kernel_size_, 1, 1);
return ggml_mul(ctx->ggml_ctx, v, ggml_div(ctx->ggml_ctx, g, norm));
}
public:
Wav2Vec2PositionalConvEmbedding(const Wav2Vec2Config& config)
: embed_dim_(config.embed_dim) {
GGML_ASSERT(embed_dim_ > 0 && embed_dim_ % groups_ == 0);
}
void init_params(ggml_context* ctx, const String2TensorStorage& tensor_storage_map = {}, const std::string prefix = "") override {
bool legacy = tensor_storage_map.count(prefix + "conv.weight_g") > 0;
weight_g_name_ = legacy ? "conv.weight_g" : "conv.parametrizations.weight.original0";
weight_v_name_ = legacy ? "conv.weight_v" : "conv.parametrizations.weight.original1";
auto g = tensor_storage_map.find(prefix + weight_g_name_);
auto v = tensor_storage_map.find(prefix + weight_v_name_);
GGML_ASSERT(g != tensor_storage_map.end() && v != tensor_storage_map.end());
GGML_ASSERT(g->second.ne[0] == kernel_size_ && g->second.ne[1] == 1 && g->second.ne[2] == 1 && g->second.ne[3] == 1);
GGML_ASSERT(v->second.ne[0] == kernel_size_ && v->second.ne[1] == embed_dim_ / groups_ && v->second.ne[2] == embed_dim_ && v->second.ne[3] == 1);
params[weight_g_name_] = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, kernel_size_, 1, 1);
params[weight_v_name_] = ggml_new_tensor_3d(ctx, get_type(prefix + weight_v_name_, tensor_storage_map, GGML_TYPE_F16),
kernel_size_, embed_dim_ / groups_, embed_dim_);
if (tensor_storage_map.count(prefix + "conv.bias") > 0) {
params["conv.bias"] = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, embed_dim_);
}
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) override {
auto w = weight(ctx);
auto b = params.count("conv.bias") > 0 ? params["conv.bias"] : nullptr;
x = ggml_cont(ctx->ggml_ctx, ggml_permute(ctx->ggml_ctx, x, 1, 0, 2, 3));
x = ggml_ext_conv_1d(ctx->ggml_ctx, x, w, b, 1, kernel_size_ / 2, 1, groups_, true);
// Apply GELU out of place before cropping to keep graph buffer reuse safe.
x = ggml_gelu_erf(ctx->ggml_ctx, x);
x = ggml_view_3d(ctx->ggml_ctx, x, x->ne[0] - 1, x->ne[1], x->ne[2], x->nb[1], x->nb[2], 0);
return ggml_permute(ctx->ggml_ctx, x, 1, 0, 2, 3);
}
};
struct Wav2Vec2FeedForward : public UnaryBlock {
Wav2Vec2FeedForward(const Wav2Vec2Config& config) {
blocks["intermediate_dense"] = std::make_shared<Linear>(config.embed_dim, config.embed_dim * 4);
blocks["output_dense"] = std::make_shared<Linear>(config.embed_dim * 4, config.embed_dim);
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) {
auto intermediate_dense = std::dynamic_pointer_cast<Linear>(blocks["intermediate_dense"]);
auto output_dense = std::dynamic_pointer_cast<Linear>(blocks["output_dense"]);
x = intermediate_dense->forward(ctx, x);
x = ggml_ext_gelu(ctx->ggml_ctx, x, true);
x = output_dense->forward(ctx, x);
return x;
}
};
struct Wav2Vec2EncoderLayer : public UnaryBlock {
bool do_stable_layer_norm;
Wav2Vec2EncoderLayer(const Wav2Vec2Config& config)
: do_stable_layer_norm(config.do_stable_layer_norm) {
blocks["attention"] = std::make_shared<MultiheadAttention>(config.embed_dim, config.num_heads, true, true);
blocks["layer_norm"] = std::make_shared<LayerNorm>(config.embed_dim);
blocks["feed_forward"] = std::make_shared<Wav2Vec2FeedForward>(config);
blocks["final_layer_norm"] = std::make_shared<LayerNorm>(config.embed_dim);
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) {
auto attention = std::dynamic_pointer_cast<MultiheadAttention>(blocks["attention"]);
auto layer_norm = std::dynamic_pointer_cast<LayerNorm>(blocks["layer_norm"]);
auto feed_forward = std::dynamic_pointer_cast<Wav2Vec2FeedForward>(blocks["feed_forward"]);
auto final_layer_norm = std::dynamic_pointer_cast<LayerNorm>(blocks["final_layer_norm"]);
ggml_tensor* residual = x;
if (do_stable_layer_norm) {
x = layer_norm->forward(ctx, x);
x = attention->forward(ctx, x);
x = ggml_add(ctx->ggml_ctx, residual, x);
x = ggml_add(ctx->ggml_ctx, x, feed_forward->forward(ctx, final_layer_norm->forward(ctx, x)));
} else {
x = attention->forward(ctx, x);
x = ggml_add(ctx->ggml_ctx, residual, x);
x = layer_norm->forward(ctx, x);
x = final_layer_norm->forward(ctx, ggml_add(ctx->ggml_ctx, x, feed_forward->forward(ctx, x)));
}
return x;
}
};
struct Wav2Vec2Encoder : public GGMLBlock {
int num_layers;
bool do_stable_layer_norm;
Wav2Vec2Encoder(const Wav2Vec2Config& config)
: num_layers(config.num_layers), do_stable_layer_norm(config.do_stable_layer_norm) {
blocks["pos_conv_embed"] = std::make_shared<Wav2Vec2PositionalConvEmbedding>(config);
for (int i = 0; i < config.num_layers; ++i) {
blocks["layers." + std::to_string(i)] = std::make_shared<Wav2Vec2EncoderLayer>(config);
}
blocks["layer_norm"] = std::make_shared<LayerNorm>(config.embed_dim);
}
// For N == 1, all_layers stacks pre-layer states and the final state as [embed_dim, L, num_layers + 1].
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x, ggml_tensor** all_layers = nullptr) {
auto pos_conv_embed = std::dynamic_pointer_cast<Wav2Vec2PositionalConvEmbedding>(blocks["pos_conv_embed"]);
auto layer_norm = std::dynamic_pointer_cast<LayerNorm>(blocks["layer_norm"]);
std::vector<ggml_tensor*> collected;
if (all_layers != nullptr) {
collected.reserve(num_layers + 1);
}
x = ggml_add(ctx->ggml_ctx, x, pos_conv_embed->forward(ctx, x));
if (!do_stable_layer_norm) {
x = layer_norm->forward(ctx, x);
}
for (int i = 0; i < num_layers; ++i) {
if (all_layers != nullptr) {
collected.push_back(x);
}
auto layer = std::dynamic_pointer_cast<Wav2Vec2EncoderLayer>(blocks["layers." + std::to_string(i)]);
x = layer->forward(ctx, x);
}
if (do_stable_layer_norm) {
x = layer_norm->forward(ctx, x);
}
if (all_layers != nullptr) {
collected.push_back(x);
ggml_tensor* stack = collected[0];
for (size_t i = 1; i < collected.size(); ++i) {
stack = ggml_concat(ctx->ggml_ctx, stack, collected[i], 2);
}
*all_layers = stack;
}
return x;
}
};
struct Wav2Vec2Model : public GGMLBlock {
Wav2Vec2Config config;
Wav2Vec2Model() = default;
Wav2Vec2Model(const Wav2Vec2Config& config_)
: config(config_) {
blocks["feature_extractor"] = std::make_shared<Wav2Vec2FeatureEncoder>(config);
blocks["feature_projection"] = std::make_shared<Wav2Vec2FeatureProjection>(config);
blocks["encoder"] = std::make_shared<Wav2Vec2Encoder>(config);
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x, ggml_tensor** all_layers = nullptr) {
auto feature_extractor = std::dynamic_pointer_cast<Wav2Vec2FeatureEncoder>(blocks["feature_extractor"]);
auto feature_projection = std::dynamic_pointer_cast<Wav2Vec2FeatureProjection>(blocks["feature_projection"]);
auto encoder = std::dynamic_pointer_cast<Wav2Vec2Encoder>(blocks["encoder"]);
x = feature_extractor->forward(ctx, x);
x = feature_projection->forward(ctx, x);
x = encoder->forward(ctx, x, all_layers);
return x;
}
};
class Wav2Vec2ModelRunner : public GGMLRunner {
private:
Wav2Vec2Config config;
public:
Wav2Vec2Model model;
std::string weight_prefix;
Wav2Vec2ModelRunner(ggml_backend_t backend,
const String2TensorStorage& tensor_storage_map = {},
const std::string prefix = "wav2vec2.",
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr)
: GGMLRunner(backend, weight_manager),
config(Wav2Vec2Config::detect_from_weights(tensor_storage_map, prefix)),
model(config),
weight_prefix(prefix) {
// GGMLBlock appends its own separator; loader prefixes already include one.
std::string block_prefix = weight_prefix;
if (!block_prefix.empty() && block_prefix.back() == '.') {
block_prefix.pop_back();
}
model.init(params_ctx, tensor_storage_map, block_prefix);
LOG_INFO("%s", get_desc().c_str());
}
std::string get_desc() override {
return "wav2vec2";
}
void get_param_tensors(std::map<std::string, ggml_tensor*>& tensors) {
std::string block_prefix = weight_prefix;
if (!block_prefix.empty() && block_prefix.back() == '.') {
block_prefix.pop_back();
}
model.get_param_tensors(tensors, block_prefix);
}
ggml_cgraph* build_graph(const sd::Tensor<float>& waveform_tensor) {
ggml_cgraph* gf = ggml_new_graph(compute_ctx);
ggml_tensor* waveform = make_input(waveform_tensor);
auto runner_ctx = get_context();
ggml_tensor* all_layers = nullptr;
model.forward(&runner_ctx, waveform, &all_layers);
GGML_ASSERT(all_layers != nullptr);
ggml_build_forward_expand(gf, all_layers);
return gf;
}
sd::Tensor<float> compute(const int n_threads, const std::vector<float>& mono_waveform) {
GGML_ASSERT(!mono_waveform.empty());
const int64_t num_samples = (int64_t)mono_waveform.size();
sd::Tensor<float> waveform({num_samples, 1, 1});
std::copy(mono_waveform.begin(), mono_waveform.end(), waveform.data());
normalize(waveform.data(), num_samples);
auto get_graph = [&]() -> ggml_cgraph* {
return build_graph(waveform);
};
return take_or_empty(GGMLRunner::compute(get_graph, n_threads, true));
}
private:
static void normalize(float* x, int64_t n) {
double mean = 0.0;
for (int64_t i = 0; i < n; ++i) {
mean += x[i];
}
mean /= n;
double var = 0.0;
for (int64_t i = 0; i < n; ++i) {
const double d = x[i] - mean;
var += d * d;
}
var /= n;
const float scale = (float)(1.0 / std::sqrt(var + 1e-7));
for (int64_t i = 0; i < n; ++i) {
x[i] = (float)((x[i] - mean) * scale);
}
}
};
} // namespace Wav2Vec2
#endif // __SD_MODEL_AUDIO_WAV2VEC2_HPP__
+2 -2
View File
@@ -380,14 +380,14 @@ public:
if (xtra_dim) {
context->ne[0] = 320; // reset dim to orig
}
x = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, q, k, v, n_head, nullptr, false, ctx->flash_attn_enabled); // [N, n_token, inner_dim]
x = ggml_ext_attention_ext(ctx, q, k, v, n_head, nullptr, false, ctx->flash_attn_enabled); // [N, n_token, inner_dim]
if (has_ip && ctx->ip_context != nullptr && ctx->ip_scale != 0.0f) {
auto to_k_ip = std::dynamic_pointer_cast<Linear>(blocks["to_k_ip"]);
auto to_v_ip = std::dynamic_pointer_cast<Linear>(blocks["to_v_ip"]);
auto k_ip = to_k_ip->forward(ctx, ctx->ip_context);
auto v_ip = to_v_ip->forward(ctx, ctx->ip_context);
auto x_ip = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, q, k_ip, v_ip, n_head, nullptr, false, ctx->flash_attn_enabled);
auto x_ip = ggml_ext_attention_ext(ctx, q, k_ip, v_ip, n_head, nullptr, false, ctx->flash_attn_enabled);
x = ggml_add(ctx->ggml_ctx, x, ggml_scale(ctx->ggml_ctx, x_ip, ctx->ip_scale));
}
+64 -3
View File
@@ -206,7 +206,9 @@ public:
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) override {
ggml_tensor* w = params["weight"];
const float scale = ctx->linear_scale > 0.f ? ctx->linear_scale : this->scale;
ggml_tensor* weight_scale = has_weight_scale ? params["weight_scale"] : nullptr;
#ifndef SD_USE_UPSTREAM_GGML
if (w->type == GGML_TYPE_F8_E4M3 || w->type == GGML_TYPE_F8_E5M2) {
bool supports_fp8_matmul = false;
if (ctx->backend != nullptr) {
@@ -220,6 +222,7 @@ public:
w = ggml_cast(ctx->ggml_ctx, w, GGML_TYPE_BF16);
}
}
#endif
ggml_tensor* b = nullptr;
if (bias) {
b = params["bias"];
@@ -237,6 +240,7 @@ public:
if (ctx->weight_adapter && b != nullptr) {
b = ctx->weight_adapter->patch_weight(ctx->ggml_ctx, ctx->backend, b, prefix + "bias");
}
#ifndef SD_USE_UPSTREAM_GGML
if (int8_convrot && scale == 1.f) {
const auto cache_key = std::make_pair(x, int8_convrot_group_size);
auto cached = ctx->int8_convrot_cache.find(cache_key);
@@ -247,6 +251,7 @@ public:
x = cached->second;
}
}
#endif
out = ggml_ext_linear_i8_tensorwise(ctx->ggml_ctx,
x,
w,
@@ -309,6 +314,7 @@ public:
__STATIC_INLINE__ bool support_get_rows(ggml_type wtype) {
switch (wtype) {
case GGML_TYPE_F16:
case GGML_TYPE_BF16:
case GGML_TYPE_Q8_0:
case GGML_TYPE_Q5_1:
case GGML_TYPE_Q5_0:
@@ -366,6 +372,61 @@ public:
}
};
class Conv1d : public UnaryBlock {
protected:
int64_t in_channels;
int64_t out_channels;
int64_t groups;
int kernel_size;
int stride;
int padding;
int dilation;
bool bias;
bool force_prec_f32;
void init_params(ggml_context* ctx, const String2TensorStorage& tensor_storage_map = {}, const std::string prefix = "") override {
ggml_type wtype = get_type(prefix + "weight", tensor_storage_map, GGML_TYPE_F16);
params["weight"] = ggml_new_tensor_3d(ctx, wtype, kernel_size, in_channels / groups, out_channels);
if (bias) {
params["bias"] = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, out_channels);
}
}
public:
Conv1d(int64_t in_channels,
int64_t out_channels,
int kernel_size,
int stride = 1,
int padding = 0,
int dilation = 1,
int64_t groups = 1,
bool bias = true,
bool force_prec_f32 = false)
: in_channels(in_channels),
out_channels(out_channels),
groups(groups),
kernel_size(kernel_size),
stride(stride),
padding(padding),
dilation(dilation),
bias(bias),
force_prec_f32(force_prec_f32) {
GGML_ASSERT(in_channels > 0 && out_channels > 0 && groups > 0);
GGML_ASSERT(in_channels % groups == 0 && out_channels % groups == 0);
GGML_ASSERT(kernel_size > 0 && stride > 0 && padding >= 0 && dilation > 0);
}
std::string get_desc() override {
return "Conv1d";
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) override {
GGML_ASSERT(x->ne[1] == in_channels);
return ggml_ext_conv_1d(ctx->ggml_ctx, x, params["weight"], bias ? params["bias"] : nullptr,
stride, padding, dilation, groups, force_prec_f32);
}
};
class Conv2d : public UnaryBlock {
protected:
int64_t in_channels;
@@ -671,7 +732,7 @@ public:
std::get<2>(stride), std::get<1>(stride), std::get<0>(stride),
std::get<2>(padding), std::get<1>(padding), std::get<0>(padding),
std::get<2>(dilation), std::get<1>(dilation), std::get<0>(dilation),
force_prec_f32);
force_prec_f32, ctx->conv3d_direct_enabled);
}
};
@@ -764,7 +825,7 @@ public:
b = ctx->weight_adapter->patch_weight(ctx->ggml_ctx, ctx->backend, b, prefix + "bias");
}
}
return ggml_ext_group_norm(ctx->ggml_ctx, x, w, b, num_groups);
return ggml_ext_group_norm(ctx->ggml_ctx, x, w, b, num_groups, eps);
}
};
@@ -869,7 +930,7 @@ public:
v = v_proj->forward(ctx, x);
}
x = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, q, k, v, n_head, mask, false); // [N, n_token, embed_dim]
x = ggml_ext_attention_ext(ctx, q, k, v, n_head, mask, false); // [N, n_token, embed_dim]
x = out_proj->forward(ctx, x); // [N, n_token, embed_dim]
return x;
+4 -3
View File
@@ -818,8 +818,9 @@ namespace Rope {
int pw,
int bs,
int theta,
const std::vector<int>& axes_dim) {
std::vector<std::vector<float>> ids = gen_vid_ids(t, h, w, pt, ph, pw, bs);
const std::vector<int>& axes_dim,
int t_offset = 0) {
std::vector<std::vector<float>> ids = gen_vid_ids(t, h, w, pt, ph, pw, bs, t_offset);
return embed_nd(ids, bs, static_cast<float>(theta), axes_dim);
}
@@ -1024,7 +1025,7 @@ namespace Rope {
q = apply_rope(ctx->ggml_ctx, q, pe, rope_interleaved); // [N*n_head, L, d_head]
k = apply_rope(ctx->ggml_ctx, k, pe, rope_interleaved); // [N*n_head, L, d_head]
auto x = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, q, k, v, n_head, mask, true, ctx->flash_attn_enabled, kv_scale); // [N, L, n_head*d_head]
auto x = ggml_ext_attention_ext(ctx, q, k, v, n_head, mask, true, ctx->flash_attn_enabled, kv_scale); // [N, L, n_head*d_head]
return x;
}
}; // namespace Rope
+2 -4
View File
@@ -237,8 +237,7 @@ namespace Anima {
}
auto q_rope = Rope::apply_rope(ctx->ggml_ctx, q4, pe_q, false);
auto k_rope = Rope::apply_rope(ctx->ggml_ctx, k4, pe_k, false);
attn_out = ggml_ext_attention_ext(ctx->ggml_ctx,
ctx->backend,
attn_out = ggml_ext_attention_ext(ctx,
q_rope,
k_rope,
v4,
@@ -249,8 +248,7 @@ namespace Anima {
} else {
auto q_flat = ggml_reshape_3d(ctx->ggml_ctx, q4, head_dim * num_heads, L_q, N);
auto k_flat = ggml_reshape_3d(ctx->ggml_ctx, k4, head_dim * num_heads, L_k, N);
attn_out = ggml_ext_attention_ext(ctx->ggml_ctx,
ctx->backend,
attn_out = ggml_ext_attention_ext(ctx,
q_flat,
k_flat,
v,
+1 -1
View File
@@ -61,7 +61,7 @@ namespace AnimateDiff {
auto k = to_k->forward(ctx, x_pe);
auto v = to_v->forward(ctx, x_pe);
auto a = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, q, k, v, (int)num_heads, nullptr, false);
auto a = ggml_ext_attention_ext(ctx, q, k, v, (int)num_heads, nullptr, false);
return to_out->forward(ctx, a);
}
};
+1 -1
View File
@@ -183,7 +183,7 @@ namespace ErnieImage {
k = ggml_cont(ctx->ggml_ctx, ggml_permute(ctx->ggml_ctx, k, 0, 2, 1, 3)); // [N, heads, S, head_dim]
k = ggml_reshape_3d(ctx->ggml_ctx, k, k->ne[0], k->ne[1], k->ne[2] * k->ne[3]);
x = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, q, k, v, num_heads, attention_mask, true, ctx->flash_attn_enabled); // [N, S, hidden_size]
x = ggml_ext_attention_ext(ctx, q, k, v, num_heads, attention_mask, true, ctx->flash_attn_enabled); // [N, S, hidden_size]
x = to_out_0->forward(ctx, x);
return x;
}
+26 -6
View File
@@ -484,13 +484,19 @@ namespace HiDreamO1 {
};
struct HiDreamO1Conditioner : public Conditioner {
Qwen2Tokenizer tokenizer;
std::shared_ptr<Tokenizer> tokenizer;
std::shared_ptr<HiDreamO1VisionRunner> vision_runner;
HiDreamO1Conditioner(ggml_backend_t backend,
const String2TensorStorage& tensor_storage_map = {},
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr)
: vision_runner(std::make_shared<HiDreamO1VisionRunner>(backend, tensor_storage_map, "model.visual", weight_manager)) {}
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr,
const TokenizerConfig& tokenizers = {})
: vision_runner(std::make_shared<HiDreamO1VisionRunner>(backend, tensor_storage_map, "model.visual", weight_manager)) {
tokenizer = tokenizers.create(TokenizerConfig::MAIN, HiDreamO1Config::detect_from_weights(tensor_storage_map, "").llm.vocab_size, 151643);
if (!tokenizer) {
tokenizer = std::make_shared<Qwen2Tokenizer>();
}
}
void get_param_tensors(std::map<std::string, ggml_tensor*>& tensors) override {
vision_runner->get_param_tensors(tensors);
@@ -504,6 +510,10 @@ namespace HiDreamO1 {
vision_runner->set_flash_attention_enabled(enabled);
}
void set_scale_overrides(float linear_scale, float attn_scale) override {
vision_runner->set_scale_overrides(linear_scale, attn_scale);
}
void set_weight_adapter(const std::shared_ptr<WeightAdapter>& adapter) override {
vision_runner->set_weight_adapter(adapter);
}
@@ -534,7 +544,10 @@ namespace HiDreamO1 {
if (ref_images.empty()) {
prompt += conditioner_params.text;
prompt += "<|im_end|>\n<|im_start|>assistant\n<|boi_token|><|tms_token|>";
auto input_ids = tokenizer.encode(prompt, nullptr);
std::vector<int> input_ids;
if (!tokenizer->encode(prompt, input_ids, nullptr)) {
return {};
}
std::vector<int32_t> input_ids_pad = input_ids;
input_ids_pad.push_back(VISION_START_TOKEN_ID);
@@ -608,7 +621,11 @@ namespace HiDreamO1 {
auto patch_img = resized_ref * 2.0f - 1.0f;
result.c_ref_images.push_back(std::move(patch_img));
int64_t prompt_start = static_cast<int64_t>(tokenizer.encode(prompt + "<|vision_start|>", nullptr).size());
std::vector<int> prefix_tokens;
if (!tokenizer->encode(prompt + "<|vision_start|>", prefix_tokens, nullptr)) {
return {};
}
int64_t prompt_start = static_cast<int64_t>(prefix_tokens.size());
prompt += "<|vision_start|>";
prompt += repeat_special_token("<|image_pad|>", image_tokens);
prompt += "<|vision_end|>";
@@ -619,7 +636,10 @@ namespace HiDreamO1 {
prompt += conditioner_params.text;
prompt += "<|im_end|>\n<|im_start|>assistant\n<|boi_token|><|tms_token|>";
auto input_ids = tokenizer.encode(prompt, nullptr);
std::vector<int> input_ids;
if (!tokenizer->encode(prompt, input_ids, nullptr)) {
return {};
}
std::vector<int32_t> input_ids_pad = input_ids;
input_ids_pad.push_back(VISION_START_TOKEN_ID);
+1 -1
View File
@@ -54,7 +54,7 @@ namespace Hunyuan {
auto k = qkv_vec[1];
auto v = qkv_vec[2];
auto attn_out = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, q, k, v, num_heads, mask, false, ctx->flash_attn_enabled);
auto attn_out = ggml_ext_attention_ext(ctx, q, k, v, num_heads, mask, false, ctx->flash_attn_enabled);
attn_out = self_attn_proj->forward(ctx, attn_out);
// adaLN_modulation
+1 -2
View File
@@ -232,8 +232,7 @@ namespace Krea2 {
q = ggml_reshape_3d(ctx->ggml_ctx, ggml_cont(ctx->ggml_ctx, q), head_dim_ * heads, Lq, N);
k = ggml_reshape_3d(ctx->ggml_ctx, ggml_cont(ctx->ggml_ctx, k), head_dim_ * kv_heads, Lk, N);
v = ggml_reshape_3d(ctx->ggml_ctx, ggml_cont(ctx->ggml_ctx, v), head_dim_ * kv_heads, Lk, N);
return ggml_ext_attention_ext(ctx->ggml_ctx,
ctx->backend,
return ggml_ext_attention_ext(ctx,
q,
k,
v,
+1 -2
View File
@@ -709,8 +709,7 @@ namespace LTXV {
k = apply_hidden_rope(ctx->ggml_ctx, k, k_pe, heads, dim_head, rope_interleaved);
}
auto out = ggml_ext_attention_ext(ctx->ggml_ctx,
ctx->backend,
auto out = ggml_ext_attention_ext(ctx,
q,
k,
v,
+1 -2
View File
@@ -215,8 +215,7 @@ namespace MiniMaxH3 {
q = attention_layout(ctx->ggml_ctx, q);
k = attention_layout(ctx->ggml_ctx, k);
}
auto out = ggml_ext_attention_ext(ctx->ggml_ctx,
ctx->backend,
auto out = ggml_ext_attention_ext(ctx,
q,
k,
v,
+7 -7
View File
@@ -365,8 +365,8 @@ public:
ggml_tensor* forward(GGMLRunnerContext* ctx,
ggml_tensor* x) {
auto qkv = pre_attention(ctx, x);
x = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, qkv[0], qkv[1], qkv[2], num_heads, nullptr, false, ctx->flash_attn_enabled); // [N, n_token, dim]
x = post_attention(ctx, x); // [N, n_token, dim]
x = ggml_ext_attention_ext(ctx, qkv[0], qkv[1], qkv[2], num_heads, nullptr, false, ctx->flash_attn_enabled); // [N, n_token, dim]
x = post_attention(ctx, x); // [N, n_token, dim]
return x;
}
};
@@ -587,8 +587,8 @@ public:
auto qkv2 = std::get<1>(qkv_intermediates);
auto intermediates = std::get<2>(qkv_intermediates);
auto attn_out = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, qkv[0], qkv[1], qkv[2], num_heads, nullptr, false, ctx->flash_attn_enabled); // [N, n_token, dim]
auto attn2_out = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, qkv2[0], qkv2[1], qkv2[2], num_heads, nullptr, false, ctx->flash_attn_enabled); // [N, n_token, dim]
auto attn_out = ggml_ext_attention_ext(ctx, qkv[0], qkv[1], qkv[2], num_heads, nullptr, false, ctx->flash_attn_enabled); // [N, n_token, dim]
auto attn2_out = ggml_ext_attention_ext(ctx, qkv2[0], qkv2[1], qkv2[2], num_heads, nullptr, false, ctx->flash_attn_enabled); // [N, n_token, dim]
x = post_attention_x(ctx,
attn_out,
attn2_out,
@@ -604,7 +604,7 @@ public:
auto qkv = qkv_intermediates.first;
auto intermediates = qkv_intermediates.second;
auto attn_out = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, qkv[0], qkv[1], qkv[2], num_heads, nullptr, false, ctx->flash_attn_enabled); // [N, n_token, dim]
auto attn_out = ggml_ext_attention_ext(ctx, qkv[0], qkv[1], qkv[2], num_heads, nullptr, false, ctx->flash_attn_enabled); // [N, n_token, dim]
x = post_attention(ctx,
attn_out,
intermediates[0],
@@ -648,7 +648,7 @@ block_mixing(GGMLRunnerContext* ctx,
qkv.push_back(ggml_concat(ctx->ggml_ctx, context_qkv[i], x_qkv[i], 1));
}
auto attn = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, qkv[0], qkv[1], qkv[2], x_block->num_heads, nullptr, false, ctx->flash_attn_enabled); // [N, n_context + n_token, hidden_size]
auto attn = ggml_ext_attention_ext(ctx, qkv[0], qkv[1], qkv[2], x_block->num_heads, nullptr, false, ctx->flash_attn_enabled); // [N, n_context + n_token, hidden_size]
auto context_attn = ggml_view_3d(ctx->ggml_ctx,
attn,
@@ -680,7 +680,7 @@ block_mixing(GGMLRunnerContext* ctx,
}
if (x_block->self_attn) {
auto attn2 = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, x_qkv2[0], x_qkv2[1], x_qkv2[2], x_block->num_heads, nullptr, false, ctx->flash_attn_enabled); // [N, n_token, hidden_size]
auto attn2 = ggml_ext_attention_ext(ctx, x_qkv2[0], x_qkv2[1], x_qkv2[2], x_block->num_heads, nullptr, false, ctx->flash_attn_enabled); // [N, n_token, hidden_size]
x = x_block->post_attention_x(ctx,
x_attn,
+12
View File
@@ -66,9 +66,15 @@ struct AnimaDiffusionExtra {
const sd::Tensor<float>* t5_weights = nullptr;
};
struct QwenImage21DiffusionExtra {
const sd::Tensor<int32_t>* image_slots = nullptr;
};
struct WanDiffusionExtra {
const sd::Tensor<float>* vace_context = nullptr;
float vace_strength = 1.f;
// S2V audio, sd::Tensor layout: [dim, T_latent*4, layers].
const sd::Tensor<float>* audio_embed = nullptr;
};
struct HiDreamO1DiffusionExtra {
@@ -114,6 +120,10 @@ struct MiniT2IDiffusionExtra {
const sd::Tensor<float>* mask = nullptr;
};
struct SenseNovaU1DiffusionExtra {
const sd::Tensor<int32_t>* input_ids = nullptr;
};
struct HunyuanVideoDiffusionExtra {
const sd::Tensor<float>* guidance = nullptr;
const sd::Tensor<float>* byt5 = nullptr;
@@ -126,11 +136,13 @@ using DiffusionExtraParams = std::variant<std::monostate,
SkipLayerDiffusionExtra,
FluxDiffusionExtra,
AnimaDiffusionExtra,
QwenImage21DiffusionExtra,
WanDiffusionExtra,
HiDreamO1DiffusionExtra,
LTXAVDiffusionExtra,
MiniMaxH3DiffusionExtra,
MiniT2IDiffusionExtra,
SenseNovaU1DiffusionExtra,
HunyuanVideoDiffusionExtra>;
struct DiffusionParams {
+384
View File
@@ -0,0 +1,384 @@
#ifndef __SD_MODEL_DIFFUSION_QWEN_IMAGE_2_1_H__
#define __SD_MODEL_DIFFUSION_QWEN_IMAGE_2_1_H__
#include "model/diffusion/qwen_image.hpp"
namespace Qwen {
struct QwenImage21Config {
int64_t in_channels = 64;
int64_t out_channels = 64;
int64_t hidden_size = 4096;
int64_t context_dim = 4096;
int64_t head_dim = 128;
int64_t intermediate_size = 12288;
int num_layers = 32;
bool fused_mlp = false;
std::vector<int> axes_dim = {16, 56, 56};
static QwenImage21Config detect_from_weights(const String2TensorStorage& weights, const std::string& prefix) {
QwenImage21Config config;
auto find = [&](const std::string& suffix) -> const TensorStorage* {
auto it = weights.find(prefix + "." + suffix);
return it == weights.end() ? nullptr : &it->second;
};
if (auto w = find("img_in.weight")) {
config.in_channels = w->ne[0];
config.hidden_size = w->ne[1];
}
if (auto w = find("proj_out.weight")) {
config.out_channels = w->ne[1];
}
if (auto w = find("txt_in.in_layer.weight")) {
config.context_dim = w->ne[0];
}
if (auto w = find("transformer_blocks.0.attn.norm_q.weight")) {
config.head_dim = w->ne[0];
}
if (auto w = find("transformer_blocks.0.img_mlp.gate_up.weight")) {
config.intermediate_size = w->ne[1] / 2;
config.fused_mlp = true;
} else if (auto w = find("transformer_blocks.0.img_mlp.proj.weight")) {
config.intermediate_size = w->ne[1];
}
int layers = 0;
const std::string block_prefix = prefix + ".transformer_blocks.";
for (const auto& [name, _] : weights) {
if (starts_with(name, block_prefix)) {
layers = std::max(layers, atoi(name.substr(block_prefix.size()).c_str()) + 1);
}
}
if (layers > 0) {
config.num_layers = layers;
LOG_VERBOSE("qwen_image_2_1: layers = %d, hidden_size = %" PRId64 ", context_dim = %" PRId64,
layers, config.hidden_size, config.context_dim);
}
return config;
}
};
struct QwenImage21Segment {
int64_t start;
int64_t end;
int64_t context_start;
int image_index;
};
struct QwenImage21Layout {
std::vector<QwenImage21Segment> segments;
std::vector<std::vector<float>> positions;
int64_t prefix_length = 0;
static QwenImage21Layout build(int64_t text_length,
const sd::Tensor<int32_t>& image_slots,
const std::vector<std::pair<int64_t, int64_t>>& image_shapes) {
if (image_shapes.empty() || (!image_slots.empty() && image_slots.numel() != text_length)) {
throw std::runtime_error("Qwen Image 2.1: invalid image token layout");
}
QwenImage21Layout layout;
int64_t position = 0;
int next_image = 0;
auto append_image = [&](int index, int64_t context_start) {
auto [height, width] = image_shapes[index];
int64_t start = static_cast<int64_t>(layout.positions.size());
layout.segments.push_back({start, start + height * width, context_start, index});
for (int64_t h = 0; h < height; ++h) {
for (int64_t w = 0; w < width; ++w) {
layout.positions.push_back({static_cast<float>(position),
static_cast<float>(h - (height - height / 2)),
static_cast<float>(w - (width - width / 2))});
}
}
position += std::max(height, width);
};
for (int64_t i = 0; i < text_length;) {
int tag = image_slots.empty() ? 0 : image_slots[i];
int64_t begin = i++;
while (i < text_length && (image_slots.empty() ? 0 : image_slots[i]) == tag) {
++i;
}
if (tag != 0) {
if (tag != next_image + 1 || next_image + 1 >= static_cast<int>(image_shapes.size()) ||
(i - begin) * 4 != image_shapes[next_image].first * image_shapes[next_image].second) {
throw std::runtime_error("Qwen Image 2.1: vision slots and reference latents must have matching sizes");
}
append_image(next_image++, begin);
} else {
int64_t start = static_cast<int64_t>(layout.positions.size());
layout.segments.push_back({start, start + i - begin, begin, -1});
for (int64_t j = begin; j < i; ++j, ++position) {
float p = static_cast<float>(position);
layout.positions.push_back({p, p, p});
}
}
}
if (next_image + 1 != static_cast<int>(image_shapes.size())) {
throw std::runtime_error("Qwen Image 2.1: missing reference image slots");
}
layout.prefix_length = static_cast<int64_t>(layout.positions.size());
append_image(next_image, text_length);
return layout;
}
};
class QwenImage21ZeroCenterRMSNorm : public RMSNorm {
public:
using RMSNorm::RMSNorm;
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) override {
auto weight = params["weight"];
if (ctx->weight_adapter) {
weight = ctx->weight_adapter->patch_weight(ctx->ggml_ctx, ctx->backend, weight, prefix + "weight");
}
weight = ggml_scale_bias(ctx->ggml_ctx, weight, 1.f, 1.f);
return ggml_mul(ctx->ggml_ctx, ggml_rms_norm(ctx->ggml_ctx, x, eps), weight);
}
};
class QwenImage21TextProjection : public GGMLBlock {
public:
QwenImage21TextProjection(const QwenImage21Config& config) {
blocks["text_norm"] = std::make_shared<QwenImage21ZeroCenterRMSNorm>(config.context_dim, 1e-6f);
blocks["in_layer"] = std::make_shared<Linear>(config.context_dim, config.hidden_size, false);
blocks["out_layer"] = std::make_shared<Linear>(config.hidden_size, config.hidden_size, false);
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) {
x = std::dynamic_pointer_cast<QwenImage21ZeroCenterRMSNorm>(blocks["text_norm"])->forward(ctx, x);
x = std::dynamic_pointer_cast<Linear>(blocks["in_layer"])->forward(ctx, x);
x = ggml_ext_gelu(ctx->ggml_ctx, x);
return std::dynamic_pointer_cast<Linear>(blocks["out_layer"])->forward(ctx, x);
}
};
class QwenImage21Attention : public QwenImageAttention {
public:
QwenImage21Attention(const QwenImage21Config& config)
: QwenImageAttention(config.hidden_size, config.head_dim, config.hidden_size / config.head_dim, 0, 0, false, false) {
for (const auto* name : {"add_q_proj", "add_k_proj", "add_v_proj", "norm_added_q", "norm_added_k", "to_add_out"}) {
blocks.erase(name);
}
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x, ggml_tensor* pe, const std::vector<QwenImage21Segment>& segments, const std::vector<ggml_tensor*>& masks) {
int64_t heads = x->ne[0] / dim_head;
auto project = [&](const char* name) {
auto h = std::dynamic_pointer_cast<Linear>(blocks[name])->forward(ctx, x);
return ggml_reshape_4d(ctx->ggml_ctx, h, dim_head, heads, x->ne[1], x->ne[2]);
};
auto q = project("to_q");
auto k = project("to_k");
auto v = project("to_v");
q = std::dynamic_pointer_cast<RMSNorm>(blocks["norm_q"])->forward(ctx, q);
k = std::dynamic_pointer_cast<RMSNorm>(blocks["norm_k"])->forward(ctx, k);
q = Rope::apply_rope(ctx->ggml_ctx, q, pe);
k = Rope::apply_rope(ctx->ggml_ctx, k, pe);
ggml_tensor* result = nullptr;
for (size_t i = 0; i < segments.size(); ++i) {
const auto& segment = segments[i];
auto sq = ggml_ext_slice(ctx->ggml_ctx, q, 1, segment.start, segment.end);
auto sk = ggml_ext_slice(ctx->ggml_ctx, k, 1, 0, segment.end);
auto sv = ggml_ext_slice(ctx->ggml_ctx, v, 2, 0, segment.end);
auto out = ggml_ext_attention_ext(ctx, sq, sk, sv, heads, masks[i], true, ctx->flash_attn_enabled);
result = result == nullptr ? out : ggml_concat(ctx->ggml_ctx, result, out, 1);
}
auto to_out = std::dynamic_pointer_cast<Linear>(blocks["to_out.0"]);
if (sd_backend_is(ctx->backend, "Vulkan") || sd_backend_is(ctx->backend, "ROCm")) {
to_out->set_force_prec_f32(true);
}
return to_out->forward(ctx, result);
}
};
class QwenImage21TransformerBlock : public GGMLBlock {
public:
QwenImage21TransformerBlock(const QwenImage21Config& config) {
blocks["img_norm1"] = std::make_shared<LayerNorm>(config.hidden_size, 1e-6f, false);
blocks["img_norm2"] = std::make_shared<LayerNorm>(config.hidden_size, 1e-6f, false);
blocks["attn"] = std::make_shared<QwenImage21Attention>(config);
if (config.fused_mlp) {
blocks["img_mlp.gate_up"] = std::make_shared<Linear>(config.hidden_size, 2 * config.intermediate_size, false);
} else {
blocks["img_mlp.proj"] = std::make_shared<Linear>(config.hidden_size, config.intermediate_size, false);
blocks["img_mlp.gate_layer"] = std::make_shared<Linear>(config.hidden_size, config.intermediate_size, false);
}
blocks["img_mlp.out"] = std::make_shared<Linear>(config.intermediate_size, config.hidden_size, false);
}
static ggml_tensor* modulate(ggml_context* ctx, ggml_tensor* x, ggml_tensor* params, int64_t prefix_length, bool gate = false) {
auto rows = ggml_ext_chunk(ctx, params, 2, 1);
auto apply = [&](ggml_tensor* part, ggml_tensor* row) {
row = gate ? ggml_tanh(ctx, row) : ggml_scale_bias(ctx, row, 1.f, 1.f);
return ggml_mul(ctx, part, row);
};
auto target = apply(ggml_ext_slice(ctx, x, 1, prefix_length, x->ne[1]), rows[0]);
if (prefix_length == 0) {
return target;
}
auto prefix = apply(ggml_ext_slice(ctx, x, 1, 0, prefix_length), rows[1]);
return ggml_concat(ctx, prefix, target, 1);
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x, const std::vector<ggml_tensor*>& modulation, ggml_tensor* pe, const QwenImage21Layout& layout, const std::vector<ggml_tensor*>& masks) {
auto h = std::dynamic_pointer_cast<LayerNorm>(blocks["img_norm1"])->forward(ctx, x);
h = modulate(ctx->ggml_ctx, h, modulation[0], layout.prefix_length);
h = std::dynamic_pointer_cast<QwenImage21Attention>(blocks["attn"])->forward(ctx, h, pe, layout.segments, masks);
x = ggml_add(ctx->ggml_ctx, x, modulate(ctx->ggml_ctx, h, modulation[1], layout.prefix_length, true));
h = std::dynamic_pointer_cast<LayerNorm>(blocks["img_norm2"])->forward(ctx, x);
h = modulate(ctx->ggml_ctx, h, modulation[2], layout.prefix_length);
ggml_tensor* gate;
auto fused = blocks.find("img_mlp.gate_up");
if (fused != blocks.end()) {
auto gate_up = std::dynamic_pointer_cast<Linear>(fused->second)->forward(ctx, h);
auto parts = ggml_ext_chunk(ctx->ggml_ctx, gate_up, 2, 0);
gate = parts[0];
h = parts[1];
} else {
gate = std::dynamic_pointer_cast<Linear>(blocks["img_mlp.gate_layer"])->forward(ctx, h);
h = std::dynamic_pointer_cast<Linear>(blocks["img_mlp.proj"])->forward(ctx, h);
}
h = ggml_mul(ctx->ggml_ctx, h, ggml_silu(ctx->ggml_ctx, gate));
h = std::dynamic_pointer_cast<Linear>(blocks["img_mlp.out"])->forward(ctx, h);
return ggml_add(ctx->ggml_ctx, x, modulate(ctx->ggml_ctx, h, modulation[3], layout.prefix_length, true));
}
};
class QwenImage21Model : public GGMLBlock {
QwenImage21Config config;
public:
QwenImage21Model(const QwenImage21Config& config)
: config(config) {
blocks["time_text_embed.timestep_embedder"] = std::make_shared<TimestepEmbedding>(256, config.hidden_size, 0, 0, false);
blocks["txt_in"] = std::make_shared<QwenImage21TextProjection>(config);
blocks["img_in"] = std::make_shared<Linear>(config.in_channels, config.hidden_size, false);
blocks["modulation.1"] = std::make_shared<Linear>(config.hidden_size, 4 * config.hidden_size, false);
blocks["norm_out.linear"] = std::make_shared<Linear>(config.hidden_size, config.hidden_size, false);
blocks["norm_out.norm"] = std::make_shared<LayerNorm>(config.hidden_size, 1e-6f, false);
blocks["proj_out"] = std::make_shared<Linear>(config.hidden_size, config.out_channels, false);
for (int i = 0; i < config.num_layers; ++i) {
blocks["transformer_blocks." + std::to_string(i)] = std::make_shared<QwenImage21TransformerBlock>(config);
}
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x, ggml_tensor* timestep, ggml_tensor* context, const std::vector<ggml_tensor*>& refs, ggml_tensor* pe, const QwenImage21Layout& layout, const std::vector<ggml_tensor*>& masks) {
auto time = ggml_concat(ctx->ggml_ctx, timestep, ggml_ext_zeros_like(ctx->ggml_ctx, timestep), 0);
// Runtime flow timesteps already use the [0, 1000] scale.
time = ggml_ext_timestep_embedding(ctx->ggml_ctx, time, 256, 10000, 1.f);
time = std::dynamic_pointer_cast<TimestepEmbedding>(blocks["time_text_embed.timestep_embedder"])->forward(ctx, time);
time = ggml_silu(ctx->ggml_ctx, time);
auto modulation = std::dynamic_pointer_cast<Linear>(blocks["modulation.1"])->forward(ctx, time);
auto mod = ggml_ext_chunk(ctx->ggml_ctx, modulation, 4, 0);
auto text = std::dynamic_pointer_cast<QwenImage21TextProjection>(blocks["txt_in"])->forward(ctx, context);
auto img_in = std::dynamic_pointer_cast<Linear>(blocks["img_in"]);
ggml_tensor* joint = nullptr;
for (const auto& segment : layout.segments) {
ggml_tensor* h;
if (segment.image_index < 0) {
h = ggml_ext_slice(ctx->ggml_ctx, text, 1, segment.context_start,
segment.context_start + segment.end - segment.start);
} else {
auto image = segment.image_index == static_cast<int>(refs.size()) ? x : refs[segment.image_index];
h = img_in->forward(ctx, DiT::patchify(ctx->ggml_ctx, image, 1, 1));
}
joint = joint == nullptr ? h : ggml_concat(ctx->ggml_ctx, joint, h, 1);
}
sd::ggml_graph_cut::mark_graph_cut(joint, "qwen_image_2_1.prelude", "joint");
for (int i = 0; i < config.num_layers; ++i) {
auto block = std::dynamic_pointer_cast<QwenImage21TransformerBlock>(blocks["transformer_blocks." + std::to_string(i)]);
joint = block->forward(ctx, joint, mod, pe, layout, masks);
sd::ggml_graph_cut::mark_graph_cut(joint, "qwen_image_2_1.transformer_blocks." + std::to_string(i), "joint");
}
joint = ggml_ext_slice(ctx->ggml_ctx, joint, 1, layout.prefix_length, joint->ne[1]);
auto scale = std::dynamic_pointer_cast<Linear>(blocks["norm_out.linear"])->forward(ctx, ggml_ext_chunk(ctx->ggml_ctx, time, 2, 1)[0]);
joint = std::dynamic_pointer_cast<LayerNorm>(blocks["norm_out.norm"])->forward(ctx, joint);
joint = ggml_mul(ctx->ggml_ctx, joint, ggml_scale_bias(ctx->ggml_ctx, scale, 1.f, 1.f));
joint = std::dynamic_pointer_cast<Linear>(blocks["proj_out"])->forward(ctx, joint);
return DiT::unpatchify_and_crop(ctx->ggml_ctx, joint, x->ne[1], x->ne[0], 1, 1);
}
};
struct QwenImage21Runner : public DiffusionModelRunner {
QwenImage21Config config;
QwenImage21Model model;
std::vector<float> pe_data;
std::vector<sd::Tensor<float>> mask_data;
QwenImage21Runner(ggml_backend_t backend, const String2TensorStorage& weights, const std::string& prefix, std::shared_ptr<RunnerWeightManager> weight_manager = nullptr)
: DiffusionModelRunner(backend, prefix, weight_manager),
config(QwenImage21Config::detect_from_weights(weights, prefix)),
model(config) {
model.init(params_ctx, weights, prefix);
}
std::string get_desc() override { return "qwen_image_2_1"; }
void get_param_tensors(std::map<std::string, ggml_tensor*>& tensors, const std::string& prefix) override {
model.get_param_tensors(tensors, prefix);
}
sd::Tensor<float> compute(int n_threads, const DiffusionParams& inputs) override {
const auto& x = tensor_or_empty(inputs.x);
const auto& context = tensor_or_empty(inputs.context);
if (x.empty() || context.empty() || context.dim() < 2 || context.shape()[0] != config.context_dim ||
tensor_or_empty(inputs.timesteps).numel() != 1 ||
x.dim() != 4 || x.shape()[3] != 1 || x.shape()[2] != config.in_channels) {
LOG_ERROR("Qwen Image 2.1 requires an image latent and text conditioning with batch size 1");
return {};
}
static const std::vector<sd::Tensor<float>> empty_refs;
const auto& refs = inputs.ref_latents && inputs.ref_image_params.pass_to_dit ? *inputs.ref_latents : empty_refs;
std::vector<std::pair<int64_t, int64_t>> shapes;
for (const auto& ref : refs) {
if (ref.dim() != 4 || ref.shape()[2] != config.in_channels || ref.shape()[3] != 1) {
LOG_ERROR("Qwen Image 2.1: invalid reference latent shape");
return {};
}
shapes.emplace_back(ref.shape()[1], ref.shape()[0]);
}
shapes.emplace_back(x.shape()[1], x.shape()[0]);
const auto* extra = std::get_if<QwenImage21DiffusionExtra>(&inputs.extra);
QwenImage21Layout layout;
try {
layout = QwenImage21Layout::build(context.shape()[1], tensor_or_empty(extra ? extra->image_slots : nullptr), shapes);
} catch (const std::exception& error) {
LOG_ERROR("%s", error.what());
return {};
}
pe_data = Rope::embed_nd(layout.positions, 1, 10000.f, config.axes_dim);
mask_data.clear();
for (const auto& segment : layout.segments) {
sd::Tensor<float> mask;
if (segment.image_index < 0) {
mask = sd::Tensor<float>::zeros({segment.end, segment.end - segment.start});
for (int64_t q = segment.start; q < segment.end; ++q) {
for (int64_t k = q + 1; k < segment.end; ++k) {
mask[k + segment.end * (q - segment.start)] = -INFINITY;
}
}
}
mask_data.push_back(std::move(mask));
}
auto build = [&]() {
auto graph = new_graph_custom(QWEN_IMAGE_GRAPH_SIZE * 2);
auto pe = ggml_new_tensor_4d(compute_ctx, GGML_TYPE_F32, 2, 2, config.head_dim / 2, layout.positions.size());
set_backend_tensor_data(pe, pe_data.data());
std::vector<ggml_tensor*> masks, ref_inputs;
for (const auto& mask : mask_data) {
masks.push_back(mask.empty() ? nullptr : make_input(mask));
}
for (const auto& ref : refs) {
ref_inputs.push_back(make_input(ref));
}
auto ctx = get_context();
auto out = model.forward(&ctx, make_input(x), make_input(*inputs.timesteps), make_input(context),
ref_inputs, pe, layout, masks);
ggml_build_forward_expand(graph, out);
return graph;
};
return restore_trailing_singleton_dims(GGMLRunner::compute(build, n_threads, false), x.dim());
}
};
}
#endif // __SD_MODEL_DIFFUSION_QWEN_IMAGE_2_1_H__
+846
View File
@@ -0,0 +1,846 @@
#ifndef __SD_MODEL_DIFFUSION_SENSENOVA_U1_H__
#define __SD_MODEL_DIFFUSION_SENSENOVA_U1_H__
#include <algorithm>
#include <cmath>
#include <cstdint>
#include <cstdlib>
#include <memory>
#include <string>
#include <unordered_set>
#include <vector>
#include "core/ggml_extend.h"
#include "model/diffusion/dit.hpp"
#include "model/diffusion/model.hpp"
#include "model/te/llm.hpp"
#include "model_loader.h"
namespace SenseNovaU1 {
constexpr int SENSENOVA_U1_GRAPH_SIZE = 327680;
struct SenseNovaU1Config {
int64_t hidden_size = 4096;
int64_t intermediate_size = 12288;
int64_t num_layers = 42;
int64_t num_heads = 32;
int64_t num_kv_heads = 8;
int64_t head_dim = 128;
int64_t vocab_size = 151936;
int64_t max_position_embeddings = 262144;
int64_t max_position_embeddings_hw = 10000;
int64_t vision_hidden_size = 1024;
int64_t patch_size = 16;
int64_t vision_downsample_factor = 2;
int64_t in_channels = 3;
int64_t timestep_embedding_size = 256;
float rms_norm_eps = 1e-6f;
float rope_theta = 5000000.f;
float rope_theta_hw = 10000.f;
float noise_scale_base_image_seq_len = 64.f;
float noise_scale_max_value = 16.f;
float t_eps = 0.02f;
bool add_noise_scale_embedding = true;
int64_t image_token_stride() const {
return patch_size * vision_downsample_factor;
}
static SenseNovaU1Config detect_from_weights(const String2TensorStorage& tensor_storage_map,
const std::string& prefix) {
SenseNovaU1Config config;
config.num_layers = 0;
const std::string root = prefix.empty() ? "" : prefix + ".";
for (const auto& [name, tensor_storage] : tensor_storage_map) {
if (!starts_with(name, root)) {
continue;
}
if (ends_with(name, "language_model.model.embed_tokens.weight") && tensor_storage.n_dims == 2) {
config.hidden_size = tensor_storage.ne[0];
config.vocab_size = tensor_storage.ne[1];
} else if (ends_with(name, "language_model.model.layers.0.mlp.gate_proj.weight") && tensor_storage.n_dims == 2) {
config.intermediate_size = tensor_storage.ne[1];
} else if (ends_with(name, "language_model.model.layers.0.self_attn.q_proj.weight") && tensor_storage.n_dims == 2) {
config.num_heads = tensor_storage.ne[1] / config.head_dim;
} else if (ends_with(name, "language_model.model.layers.0.self_attn.k_proj.weight") && tensor_storage.n_dims == 2) {
config.num_kv_heads = tensor_storage.ne[1] / config.head_dim;
} else if (ends_with(name, "fm_modules.vision_model_mot_gen.embeddings.patch_embedding.weight") && tensor_storage.n_dims == 4) {
config.patch_size = tensor_storage.ne[0];
config.in_channels = tensor_storage.ne[2];
config.vision_hidden_size = tensor_storage.ne[3];
} else if (ends_with(name, "fm_modules.vision_model_mot_gen.embeddings.dense_embedding.weight") && tensor_storage.n_dims == 4) {
config.vision_downsample_factor = tensor_storage.ne[0];
}
const std::string layer_prefix = root + "language_model.model.layers.";
if (starts_with(name, layer_prefix)) {
const char* index_begin = name.c_str() + layer_prefix.size();
config.num_layers = std::max<int64_t>(config.num_layers, std::strtoll(index_begin, nullptr, 10) + 1);
}
}
if (config.num_layers == 0) {
config.num_layers = 42;
}
config.add_noise_scale_embedding = tensor_storage_map.find(root + "fm_modules.noise_scale_embedder.mlp.0.weight") != tensor_storage_map.end();
LOG_DEBUG("sensenova-u1.5: layers=%" PRId64 ", hidden=%" PRId64 ", intermediate=%" PRId64 ", heads=%" PRId64 ", kv_heads=%" PRId64 ", patch=%" PRId64 "x%" PRId64,
config.num_layers,
config.hidden_size,
config.intermediate_size,
config.num_heads,
config.num_kv_heads,
config.patch_size,
config.vision_downsample_factor);
return config;
}
};
class StorageConv2d : public Conv2d {
protected:
void init_params(ggml_context* ctx,
const String2TensorStorage& tensor_storage_map = {},
const std::string prefix = "") override {
this->prefix = prefix;
ggml_type wtype = get_type(prefix + "weight", tensor_storage_map, GGML_TYPE_F16);
params["weight"] = ggml_new_tensor_4d(ctx,
wtype,
kernel_size.second,
kernel_size.first,
in_channels,
out_channels);
if (bias) {
params["bias"] = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, out_channels);
}
}
public:
StorageConv2d(int64_t in_channels,
int64_t out_channels,
std::pair<int, int> kernel_size,
std::pair<int, int> stride = {1, 1},
std::pair<int, int> padding = {0, 0},
bool bias = true)
: Conv2d(in_channels,
out_channels,
kernel_size,
stride,
padding,
{1, 1},
bias) {}
};
struct TimestepEmbedder : public GGMLBlock {
int64_t frequency_embedding_size;
TimestepEmbedder(int64_t hidden_size, int64_t frequency_embedding_size = 256)
: frequency_embedding_size(frequency_embedding_size) {
blocks["mlp.0"] = std::make_shared<Linear>(frequency_embedding_size, hidden_size, true);
blocks["mlp.2"] = std::make_shared<Linear>(hidden_size, hidden_size, true);
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* timesteps) {
auto mlp_0 = std::dynamic_pointer_cast<Linear>(blocks["mlp.0"]);
auto mlp_2 = std::dynamic_pointer_cast<Linear>(blocks["mlp.2"]);
auto x = ggml_ext_timestep_embedding(ctx->ggml_ctx,
timesteps,
static_cast<int>(frequency_embedding_size),
10000,
1.f);
x = mlp_0->forward(ctx, x);
x = ggml_silu_inplace(ctx->ggml_ctx, x);
return mlp_2->forward(ctx, x);
}
};
inline ggml_tensor* apply_vision_rope(GGMLRunnerContext* ctx,
ggml_tensor* x,
ggml_tensor* position_x,
ggml_tensor* position_y,
float theta,
int max_position) {
GGML_ASSERT(x->ne[0] % 2 == 0);
// ggml_rope_ext addresses positions through ne[2]. The vision
// embeddings arrive as [hidden, tokens, batch], so add the singleton
// head axis used by the RoPE kernel: [hidden, 1, tokens, batch].
x = ggml_reshape_4d(ctx->ggml_ctx, x, x->ne[0], 1, x->ne[1], x->ne[2]);
const int64_t half = x->ne[0] / 2;
auto x_part = ggml_ext_slice(ctx->ggml_ctx, x, 0, 0, half);
auto y_part = ggml_ext_slice(ctx->ggml_ctx, x, 0, half, x->ne[0]);
x_part = ggml_rope_ext(ctx->ggml_ctx,
x_part,
position_x,
nullptr,
static_cast<int>(half),
GGML_ROPE_TYPE_NORMAL,
max_position,
theta,
1.f,
0.f,
1.f,
32.f,
1.f);
y_part = ggml_rope_ext(ctx->ggml_ctx,
y_part,
position_y,
nullptr,
static_cast<int>(half),
GGML_ROPE_TYPE_NORMAL,
max_position,
theta,
1.f,
0.f,
1.f,
32.f,
1.f);
return ggml_concat(ctx->ggml_ctx, x_part, y_part, 0);
}
struct VisionEmbeddings : public GGMLBlock {
SenseNovaU1Config config;
explicit VisionEmbeddings(const SenseNovaU1Config& config)
: config(config) {
blocks["patch_embedding"] = std::make_shared<StorageConv2d>(config.in_channels,
config.vision_hidden_size,
std::pair<int, int>{static_cast<int>(config.patch_size), static_cast<int>(config.patch_size)},
std::pair<int, int>{static_cast<int>(config.patch_size), static_cast<int>(config.patch_size)},
std::pair<int, int>{0, 0},
true);
blocks["dense_embedding"] = std::make_shared<StorageConv2d>(config.vision_hidden_size,
config.hidden_size,
std::pair<int, int>{static_cast<int>(config.vision_downsample_factor), static_cast<int>(config.vision_downsample_factor)},
std::pair<int, int>{static_cast<int>(config.vision_downsample_factor), static_cast<int>(config.vision_downsample_factor)},
std::pair<int, int>{0, 0},
true);
}
ggml_tensor* forward(GGMLRunnerContext* ctx,
ggml_tensor* image,
ggml_tensor* position_x,
ggml_tensor* position_y) {
auto patch_embedding = std::dynamic_pointer_cast<StorageConv2d>(blocks["patch_embedding"]);
auto dense_embedding = std::dynamic_pointer_cast<StorageConv2d>(blocks["dense_embedding"]);
auto x = patch_embedding->forward(ctx, image);
x = ggml_gelu_erf(ctx->ggml_ctx, x);
const int64_t grid_w = x->ne[0];
const int64_t grid_h = x->ne[1];
const int64_t batch = x->ne[3];
x = ggml_reshape_3d(ctx->ggml_ctx, x, grid_w * grid_h, x->ne[2], batch);
x = ggml_cont(ctx->ggml_ctx, ggml_permute(ctx->ggml_ctx, x, 1, 0, 2, 3));
x = apply_vision_rope(ctx,
x,
position_x,
position_y,
config.rope_theta_hw,
static_cast<int>(config.max_position_embeddings_hw));
x = ggml_reshape_4d(ctx->ggml_ctx, x, config.vision_hidden_size, grid_w, grid_h, batch);
x = ggml_cont(ctx->ggml_ctx, ggml_permute(ctx->ggml_ctx, x, 2, 0, 1, 3));
x = dense_embedding->forward(ctx, x);
const int64_t token_w = x->ne[0];
const int64_t token_h = x->ne[1];
x = ggml_reshape_3d(ctx->ggml_ctx, x, token_w * token_h, x->ne[2], x->ne[3]);
return ggml_cont(ctx->ggml_ctx, ggml_permute(ctx->ggml_ctx, x, 1, 0, 2, 3));
}
};
inline ggml_tensor* pixel_shuffle(GGMLRunnerContext* ctx,
ggml_tensor* x,
int upscale_factor) {
GGML_ASSERT(upscale_factor > 0);
const int64_t h = x->ne[1];
const int64_t w = x->ne[0];
GGML_ASSERT(x->ne[2] % (upscale_factor * upscale_factor) == 0);
x = ggml_ext_cont(ctx->ggml_ctx,
ggml_ext_torch_permute(ctx->ggml_ctx, x, 2, 0, 1, 3));
x = ggml_reshape_3d(ctx->ggml_ctx, x, x->ne[0], x->ne[1] * x->ne[2], x->ne[3]);
return DiT::unpatchify(ctx->ggml_ctx, x, h, w, upscale_factor, upscale_factor, true);
}
struct PixelDecoder : public GGMLBlock {
explicit PixelDecoder(const SenseNovaU1Config& config) {
blocks["conv1"] = std::make_shared<StorageConv2d>(config.hidden_size / 4,
1024,
std::pair<int, int>{3, 3},
std::pair<int, int>{1, 1},
std::pair<int, int>{1, 1},
true);
blocks["conv2"] = std::make_shared<StorageConv2d>(256,
192,
std::pair<int, int>{3, 3},
std::pair<int, int>{1, 1},
std::pair<int, int>{1, 1},
true);
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) {
auto conv1 = std::dynamic_pointer_cast<StorageConv2d>(blocks["conv1"]);
auto conv2 = std::dynamic_pointer_cast<StorageConv2d>(blocks["conv2"]);
x = pixel_shuffle(ctx, x, 2);
x = conv1->forward(ctx, x);
x = ggml_gelu_erf(ctx->ggml_ctx, x);
x = pixel_shuffle(ctx, x, 2);
x = conv2->forward(ctx, x);
return pixel_shuffle(ctx, x, 8);
}
};
enum class Branch {
UNDERSTANDING,
GENERATION,
};
struct Attention : public GGMLBlock {
SenseNovaU1Config config;
int layer_index;
Attention(const SenseNovaU1Config& config, int layer_index)
: config(config), layer_index(layer_index) {
blocks["q_proj"] = std::make_shared<Linear>(config.hidden_size, config.num_heads * config.head_dim, false);
blocks["k_proj"] = std::make_shared<Linear>(config.hidden_size, config.num_kv_heads * config.head_dim, false);
blocks["v_proj"] = std::make_shared<Linear>(config.hidden_size, config.num_kv_heads * config.head_dim, false);
blocks["o_proj"] = std::make_shared<Linear>(config.num_heads * config.head_dim, config.hidden_size, false);
blocks["q_proj_mot_gen"] = std::make_shared<Linear>(config.hidden_size, config.num_heads * config.head_dim, false);
blocks["k_proj_mot_gen"] = std::make_shared<Linear>(config.hidden_size, config.num_kv_heads * config.head_dim, false);
blocks["v_proj_mot_gen"] = std::make_shared<Linear>(config.hidden_size, config.num_kv_heads * config.head_dim, false);
blocks["o_proj_mot_gen"] = std::make_shared<Linear>(config.num_heads * config.head_dim, config.hidden_size, false);
const int64_t axis_dim = config.head_dim / 2;
blocks["q_norm"] = std::make_shared<LLM::LLMRMSNorm>(axis_dim, config.rms_norm_eps);
blocks["k_norm"] = std::make_shared<LLM::LLMRMSNorm>(axis_dim, config.rms_norm_eps);
blocks["q_norm_hw"] = std::make_shared<LLM::LLMRMSNorm>(axis_dim, config.rms_norm_eps);
blocks["k_norm_hw"] = std::make_shared<LLM::LLMRMSNorm>(axis_dim, config.rms_norm_eps);
blocks["q_norm_mot_gen"] = std::make_shared<LLM::LLMRMSNorm>(axis_dim, config.rms_norm_eps);
blocks["k_norm_mot_gen"] = std::make_shared<LLM::LLMRMSNorm>(axis_dim, config.rms_norm_eps);
blocks["q_norm_hw_mot_gen"] = std::make_shared<LLM::LLMRMSNorm>(axis_dim, config.rms_norm_eps);
blocks["k_norm_hw_mot_gen"] = std::make_shared<LLM::LLMRMSNorm>(axis_dim, config.rms_norm_eps);
}
ggml_tensor* apply_axis_rope(GGMLRunnerContext* ctx,
ggml_tensor* x,
ggml_tensor* positions,
int dimensions,
float theta,
int max_position) {
return ggml_rope_ext(ctx->ggml_ctx,
x,
positions,
nullptr,
dimensions,
GGML_ROPE_TYPE_NEOX,
max_position,
theta,
1.f,
0.f,
1.f,
32.f,
1.f);
}
ggml_tensor* normalize_and_rotate(GGMLRunnerContext* ctx,
ggml_tensor* x,
ggml_tensor* position_t,
ggml_tensor* position_h,
ggml_tensor* position_w,
const std::string& norm_name,
const std::string& norm_hw_name) {
const int64_t temporal_dim = config.head_dim / 2;
const int64_t spatial_dim = config.head_dim - temporal_dim;
const int64_t axis_dim = spatial_dim / 2;
auto temporal = ggml_ext_slice(ctx->ggml_ctx, x, 0, 0, temporal_dim);
auto spatial = ggml_ext_slice(ctx->ggml_ctx, x, 0, temporal_dim, config.head_dim);
temporal = std::dynamic_pointer_cast<LLM::LLMRMSNorm>(blocks[norm_name])->forward(ctx, temporal);
spatial = std::dynamic_pointer_cast<LLM::LLMRMSNorm>(blocks[norm_hw_name])->forward(ctx, spatial);
auto height = ggml_ext_slice(ctx->ggml_ctx, spatial, 0, 0, axis_dim);
auto width = ggml_ext_slice(ctx->ggml_ctx, spatial, 0, axis_dim, spatial_dim);
temporal = apply_axis_rope(ctx,
temporal,
position_t,
static_cast<int>(temporal_dim),
config.rope_theta,
static_cast<int>(config.max_position_embeddings));
height = apply_axis_rope(ctx,
height,
position_h,
static_cast<int>(axis_dim),
config.rope_theta_hw,
static_cast<int>(config.max_position_embeddings_hw));
width = apply_axis_rope(ctx,
width,
position_w,
static_cast<int>(axis_dim),
config.rope_theta_hw,
static_cast<int>(config.max_position_embeddings_hw));
return ggml_concat(ctx->ggml_ctx,
ggml_concat(ctx->ggml_ctx, temporal, height, 0),
width,
0);
}
ggml_tensor* forward(GGMLRunnerContext* ctx,
ggml_tensor* x,
ggml_tensor* position_t,
ggml_tensor* position_h,
ggml_tensor* position_w,
ggml_tensor* attention_mask,
Branch branch,
const std::string& cache_prefix) {
const bool generation = branch == Branch::GENERATION;
const std::string suffix = generation ? "_mot_gen" : "";
auto q_proj = std::dynamic_pointer_cast<Linear>(blocks["q_proj" + suffix]);
auto k_proj = std::dynamic_pointer_cast<Linear>(blocks["k_proj" + suffix]);
auto v_proj = std::dynamic_pointer_cast<Linear>(blocks["v_proj" + suffix]);
auto o_proj = std::dynamic_pointer_cast<Linear>(blocks["o_proj" + suffix]);
const int64_t n_tokens = x->ne[1];
const int64_t batch = x->ne[2];
auto q = ggml_reshape_4d(ctx->ggml_ctx,
q_proj->forward(ctx, x),
config.head_dim,
config.num_heads,
n_tokens,
batch);
auto k = ggml_reshape_4d(ctx->ggml_ctx,
k_proj->forward(ctx, x),
config.head_dim,
config.num_kv_heads,
n_tokens,
batch);
auto v = ggml_reshape_4d(ctx->ggml_ctx,
v_proj->forward(ctx, x),
config.head_dim,
config.num_kv_heads,
n_tokens,
batch);
q = normalize_and_rotate(ctx,
q,
position_t,
position_h,
position_w,
"q_norm" + suffix,
"q_norm_hw" + suffix);
k = normalize_and_rotate(ctx,
k,
position_t,
position_h,
position_w,
"k_norm" + suffix,
"k_norm_hw" + suffix);
const std::string layer_cache = cache_prefix + "." + std::to_string(layer_index);
if (generation) {
auto prefix_k = ctx->load_cache_tensor(layer_cache + ".k");
auto prefix_v = ctx->load_cache_tensor(layer_cache + ".v");
GGML_ASSERT(prefix_k != nullptr && prefix_v != nullptr);
k = ggml_concat(ctx->ggml_ctx, prefix_k, k, 2);
v = ggml_concat(ctx->ggml_ctx, prefix_v, v, 2);
} else {
// Keep dedicated graph outputs alive until the runner copies them
// into its persistent cache buffer after graph execution.
auto cache_k = ggml_dup_tensor(ctx->ggml_ctx, k);
cache_k = ggml_cpy(ctx->ggml_ctx, k, cache_k);
ggml_set_output(cache_k);
auto cache_v = ggml_dup_tensor(ctx->ggml_ctx, v);
cache_v = ggml_cpy(ctx->ggml_ctx, v, cache_v);
ggml_set_output(cache_v);
ctx->persist_cache_tensor(layer_cache + ".k", cache_k);
ctx->persist_cache_tensor(layer_cache + ".v", cache_v);
}
q = ggml_cont(ctx->ggml_ctx,
ggml_ext_torch_permute(ctx->ggml_ctx, q, 0, 2, 1, 3));
q = ggml_reshape_3d(ctx->ggml_ctx, q, q->ne[0], q->ne[1], q->ne[2] * q->ne[3]);
k = ggml_cont(ctx->ggml_ctx,
ggml_ext_torch_permute(ctx->ggml_ctx, k, 0, 2, 1, 3));
k = ggml_reshape_3d(ctx->ggml_ctx, k, k->ne[0], k->ne[1], k->ne[2] * k->ne[3]);
auto out = ggml_ext_attention_ext(ctx->ggml_ctx,
ctx->backend,
q,
k,
v,
config.num_heads,
attention_mask,
true,
ctx->flash_attn_enabled);
return o_proj->forward(ctx, out);
}
};
struct TransformerBlock : public GGMLBlock {
TransformerBlock(const SenseNovaU1Config& config, int layer_index) {
blocks["self_attn"] = std::make_shared<Attention>(config, layer_index);
blocks["mlp"] = std::make_shared<LLM::MLP>(config.hidden_size, config.intermediate_size, false);
blocks["mlp_mot_gen"] = std::make_shared<LLM::MLP>(config.hidden_size, config.intermediate_size, false);
blocks["input_layernorm"] = std::make_shared<LLM::LLMRMSNorm>(config.hidden_size, config.rms_norm_eps);
blocks["input_layernorm_mot_gen"] = std::make_shared<LLM::LLMRMSNorm>(config.hidden_size, config.rms_norm_eps);
blocks["post_attention_layernorm"] = std::make_shared<LLM::LLMRMSNorm>(config.hidden_size, config.rms_norm_eps);
blocks["post_attention_layernorm_mot_gen"] = std::make_shared<LLM::LLMRMSNorm>(config.hidden_size, config.rms_norm_eps);
}
ggml_tensor* forward(GGMLRunnerContext* ctx,
ggml_tensor* x,
ggml_tensor* position_t,
ggml_tensor* position_h,
ggml_tensor* position_w,
ggml_tensor* attention_mask,
Branch branch,
const std::string& cache_prefix) {
const bool generation = branch == Branch::GENERATION;
auto input_norm = std::dynamic_pointer_cast<LLM::LLMRMSNorm>(
blocks[generation ? "input_layernorm_mot_gen" : "input_layernorm"]);
auto post_norm = std::dynamic_pointer_cast<LLM::LLMRMSNorm>(
blocks[generation ? "post_attention_layernorm_mot_gen" : "post_attention_layernorm"]);
auto attention = std::dynamic_pointer_cast<Attention>(blocks["self_attn"]);
auto mlp = std::dynamic_pointer_cast<LLM::MLP>(blocks[generation ? "mlp_mot_gen" : "mlp"]);
auto residual = x;
x = input_norm->forward(ctx, x);
x = attention->forward(ctx,
x,
position_t,
position_h,
position_w,
attention_mask,
branch,
cache_prefix);
x = ggml_add_inplace(ctx->ggml_ctx, x, residual);
residual = x;
x = post_norm->forward(ctx, x);
x = mlp->forward(ctx, x);
return ggml_add_inplace(ctx->ggml_ctx, x, residual);
}
};
struct TextModel : public GGMLBlock {
SenseNovaU1Config config;
explicit TextModel(const SenseNovaU1Config& config)
: config(config) {
blocks["embed_tokens"] = std::make_shared<Embedding>(config.vocab_size, config.hidden_size);
for (int i = 0; i < config.num_layers; ++i) {
blocks["layers." + std::to_string(i)] = std::make_shared<TransformerBlock>(config, i);
}
blocks["norm"] = std::make_shared<LLM::LLMRMSNorm>(config.hidden_size, config.rms_norm_eps);
blocks["norm_mot_gen"] = std::make_shared<LLM::LLMRMSNorm>(config.hidden_size, config.rms_norm_eps);
}
ggml_tensor* embed(GGMLRunnerContext* ctx, ggml_tensor* input_ids) {
return std::dynamic_pointer_cast<Embedding>(blocks["embed_tokens"])->forward(ctx, input_ids);
}
ggml_tensor* forward(GGMLRunnerContext* ctx,
ggml_tensor* x,
ggml_tensor* position_t,
ggml_tensor* position_h,
ggml_tensor* position_w,
ggml_tensor* attention_mask,
Branch branch,
const std::string& cache_prefix) {
for (int i = 0; i < config.num_layers; ++i) {
auto layer = std::dynamic_pointer_cast<TransformerBlock>(blocks["layers." + std::to_string(i)]);
x = layer->forward(ctx,
x,
position_t,
position_h,
position_w,
attention_mask,
branch,
cache_prefix);
}
auto norm = std::dynamic_pointer_cast<LLM::LLMRMSNorm>(
blocks[branch == Branch::GENERATION ? "norm_mot_gen" : "norm"]);
return norm->forward(ctx, x);
}
};
struct SenseNovaU1Model : public GGMLBlock {
SenseNovaU1Config config;
explicit SenseNovaU1Model(const SenseNovaU1Config& config)
: config(config) {
blocks["language_model.model"] = std::make_shared<TextModel>(config);
blocks["fm_modules.vision_model_mot_gen.embeddings"] = std::make_shared<VisionEmbeddings>(config);
blocks["fm_modules.timestep_embedder"] = std::make_shared<TimestepEmbedder>(config.hidden_size,
config.timestep_embedding_size);
if (config.add_noise_scale_embedding) {
blocks["fm_modules.noise_scale_embedder"] = std::make_shared<TimestepEmbedder>(config.hidden_size,
config.timestep_embedding_size);
}
blocks["fm_modules.fm_head"] = std::make_shared<PixelDecoder>(config);
}
std::shared_ptr<TextModel> text_model() {
return std::dynamic_pointer_cast<TextModel>(blocks["language_model.model"]);
}
std::shared_ptr<VisionEmbeddings> vision_embeddings() {
return std::dynamic_pointer_cast<VisionEmbeddings>(blocks["fm_modules.vision_model_mot_gen.embeddings"]);
}
std::shared_ptr<TimestepEmbedder> timestep_embedder() {
return std::dynamic_pointer_cast<TimestepEmbedder>(blocks["fm_modules.timestep_embedder"]);
}
std::shared_ptr<TimestepEmbedder> noise_scale_embedder() {
if (!config.add_noise_scale_embedding) {
return nullptr;
}
return std::dynamic_pointer_cast<TimestepEmbedder>(blocks["fm_modules.noise_scale_embedder"]);
}
std::shared_ptr<PixelDecoder> pixel_decoder() {
return std::dynamic_pointer_cast<PixelDecoder>(blocks["fm_modules.fm_head"]);
}
};
struct SenseNovaU1Runner : public DiffusionModelRunner {
SenseNovaU1Config config;
SenseNovaU1Model model;
std::unordered_set<uint64_t> cached_prefix_hashes;
std::vector<int32_t> position_t_vec;
std::vector<int32_t> position_h_vec;
std::vector<int32_t> position_w_vec;
std::vector<float> attention_mask_vec;
std::vector<float> noise_scale_vec;
SenseNovaU1Runner(ggml_backend_t backend,
const String2TensorStorage& tensor_storage_map = {},
const std::string& prefix = "",
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr)
: DiffusionModelRunner(backend, prefix, weight_manager),
config(SenseNovaU1Config::detect_from_weights(tensor_storage_map, prefix)),
model(config) {
model.init(params_ctx, tensor_storage_map, prefix);
}
std::string get_desc() override {
return "SenseNova U1.5";
}
void get_param_tensors(std::map<std::string, ggml_tensor*>& tensors,
const std::string& prefix) override {
model.get_param_tensors(tensors, prefix);
}
static uint64_t hash_input_ids(const sd::Tensor<int32_t>& input_ids) {
uint64_t hash = 1469598103934665603ULL;
for (int32_t token : input_ids.values()) {
uint32_t value = static_cast<uint32_t>(token);
for (int byte = 0; byte < 4; ++byte) {
hash ^= static_cast<uint8_t>(value & 0xffU);
hash *= 1099511628211ULL;
value >>= 8;
}
}
hash ^= static_cast<uint64_t>(input_ids.numel());
hash *= 1099511628211ULL;
return hash;
}
static std::string cache_prefix(uint64_t hash) {
return "snu15." + std::to_string(hash);
}
ggml_tensor* make_position_tensor(const std::vector<int32_t>& values,
const std::string& name) {
auto tensor = ggml_new_tensor_1d(compute_ctx, GGML_TYPE_I32, values.size());
ggml_set_name(tensor, name.c_str());
set_backend_tensor_data(tensor, values.data());
return tensor;
}
ggml_cgraph* build_prefix_graph(const sd::Tensor<int32_t>& input_ids_tensor,
const std::string& prefix_cache) {
ggml_cgraph* graph = new_graph_custom(SENSENOVA_U1_GRAPH_SIZE);
ggml_tensor* ids = make_input(input_ids_tensor);
const int64_t length = input_ids_tensor.numel();
position_t_vec.resize(length);
position_h_vec.assign(length, 0);
position_w_vec.assign(length, 0);
for (int64_t i = 0; i < length; ++i) {
position_t_vec[i] = static_cast<int32_t>(i);
}
auto position_t = make_position_tensor(position_t_vec, "snu15.prefix.position_t");
auto position_h = make_position_tensor(position_h_vec, "snu15.prefix.position_h");
auto position_w = make_position_tensor(position_w_vec, "snu15.prefix.position_w");
attention_mask_vec.assign(static_cast<size_t>(length * length), 0.f);
for (int64_t query = 0; query < length; ++query) {
for (int64_t key = query + 1; key < length; ++key) {
attention_mask_vec[static_cast<size_t>(query * length + key)] = -INFINITY;
}
}
auto attention_mask = ggml_new_tensor_2d(compute_ctx,
GGML_TYPE_F32,
length,
length);
ggml_set_name(attention_mask, "snu15.prefix.attention_mask");
set_backend_tensor_data(attention_mask, attention_mask_vec.data());
auto runner_ctx = get_context();
auto text_model = model.text_model();
auto hidden = text_model->embed(&runner_ctx, ids);
hidden = text_model->forward(&runner_ctx,
hidden,
position_t,
position_h,
position_w,
attention_mask,
Branch::UNDERSTANDING,
prefix_cache);
ggml_build_forward_expand(graph, hidden);
return graph;
}
bool ensure_prefix_cache(int n_threads,
const sd::Tensor<int32_t>& input_ids,
std::string* prefix_cache) {
const uint64_t hash = hash_input_ids(input_ids);
*prefix_cache = cache_prefix(hash);
if (cached_prefix_hashes.find(hash) != cached_prefix_hashes.end() &&
get_cache_tensor_by_name(*prefix_cache + ".0.k") != nullptr) {
return true;
}
if (cached_prefix_hashes.size() >= 2) {
free_cache_ctx_and_buffer();
cached_prefix_hashes.clear();
}
auto get_graph = [&]() {
return build_prefix_graph(input_ids, *prefix_cache);
};
auto result = GGMLRunner::compute(get_graph, n_threads, false, true);
if (!result.has_value()) {
LOG_ERROR("SenseNova U1.5 prefix cache computation failed");
return false;
}
cached_prefix_hashes.insert(hash);
return true;
}
ggml_cgraph* build_graph(const sd::Tensor<float>& x_tensor,
const sd::Tensor<float>& timestep_tensor,
const std::string& prefix_cache,
int64_t prefix_length) {
ggml_cgraph* graph = new_graph_custom(SENSENOVA_U1_GRAPH_SIZE);
ggml_tensor* x = make_input(x_tensor);
ggml_tensor* t = make_input(timestep_tensor);
GGML_ASSERT(x->ne[3] == 1);
GGML_ASSERT(x->ne[0] % config.image_token_stride() == 0);
GGML_ASSERT(x->ne[1] % config.image_token_stride() == 0);
const int64_t grid_w = x->ne[0] / config.patch_size;
const int64_t grid_h = x->ne[1] / config.patch_size;
const int64_t token_w = grid_w / config.vision_downsample_factor;
const int64_t token_h = grid_h / config.vision_downsample_factor;
const int64_t tokens = token_w * token_h;
position_h_vec.resize(grid_w * grid_h);
position_w_vec.resize(grid_w * grid_h);
for (int64_t index = 0; index < grid_w * grid_h; ++index) {
position_h_vec[index] = static_cast<int32_t>(index / grid_w);
position_w_vec[index] = static_cast<int32_t>(index % grid_w);
}
auto vision_position_x = make_position_tensor(position_w_vec, "snu15.vision.position_x");
auto vision_position_y = make_position_tensor(position_h_vec, "snu15.vision.position_y");
auto runner_ctx = get_context();
auto hidden = model.vision_embeddings()->forward(&runner_ctx,
x,
vision_position_x,
vision_position_y);
auto time_embedding = model.timestep_embedder()->forward(&runner_ctx, t);
time_embedding = ggml_reshape_3d(compute_ctx, time_embedding, config.hidden_size, 1, 1);
hidden = ggml_add(compute_ctx, hidden, time_embedding);
if (config.add_noise_scale_embedding) {
const float image_tokens = static_cast<float>(tokens);
const float noise_scale = std::min(config.noise_scale_max_value,
std::sqrt(image_tokens / config.noise_scale_base_image_seq_len));
noise_scale_vec = {noise_scale / config.noise_scale_max_value};
auto noise_scale_tensor = ggml_new_tensor_1d(compute_ctx, GGML_TYPE_F32, 1);
ggml_set_name(noise_scale_tensor, "snu15.noise_scale");
set_backend_tensor_data(noise_scale_tensor, noise_scale_vec.data());
auto noise_embedding = model.noise_scale_embedder()->forward(&runner_ctx, noise_scale_tensor);
noise_embedding = ggml_reshape_3d(compute_ctx, noise_embedding, config.hidden_size, 1, 1);
hidden = ggml_add(compute_ctx, hidden, noise_embedding);
}
position_t_vec.assign(tokens, static_cast<int32_t>(prefix_length));
position_h_vec.resize(tokens);
position_w_vec.resize(tokens);
for (int64_t index = 0; index < tokens; ++index) {
position_h_vec[index] = static_cast<int32_t>(index / token_w);
position_w_vec[index] = static_cast<int32_t>(index % token_w);
}
auto position_t = make_position_tensor(position_t_vec, "snu15.image.position_t");
auto position_h = make_position_tensor(position_h_vec, "snu15.image.position_h");
auto position_w = make_position_tensor(position_w_vec, "snu15.image.position_w");
hidden = model.text_model()->forward(&runner_ctx,
hidden,
position_t,
position_h,
position_w,
nullptr,
Branch::GENERATION,
prefix_cache);
hidden = ggml_reshape_4d(compute_ctx,
hidden,
config.hidden_size,
token_w,
token_h,
x->ne[3]);
hidden = ggml_cont(compute_ctx, ggml_permute(compute_ctx, hidden, 2, 0, 1, 3));
auto x_prediction = model.pixel_decoder()->forward(&runner_ctx, hidden);
const float timestep = timestep_tensor.values()[0];
const float denom = std::max(1.f - timestep, config.t_eps);
auto velocity = ggml_scale(compute_ctx,
ggml_sub(compute_ctx, x_prediction, x),
1.f / denom);
ggml_build_forward_expand(graph, velocity);
return graph;
}
sd::Tensor<float> compute(int n_threads,
const sd::Tensor<float>& x,
const sd::Tensor<float>& timestep,
const sd::Tensor<int32_t>& input_ids) {
std::string prefix_cache;
if (!ensure_prefix_cache(n_threads, input_ids, &prefix_cache)) {
return {};
}
auto get_graph = [&]() {
return build_graph(x, timestep, prefix_cache, input_ids.numel());
};
return restore_trailing_singleton_dims(
GGMLRunner::compute(get_graph, n_threads, false),
x.dim());
}
sd::Tensor<float> compute(int n_threads,
const DiffusionParams& diffusion_params) override {
GGML_ASSERT(diffusion_params.x != nullptr);
GGML_ASSERT(diffusion_params.timesteps != nullptr);
const auto* extra = diffusion_extra_as<SenseNovaU1DiffusionExtra>(diffusion_params);
GGML_ASSERT(extra->input_ids != nullptr);
return compute(n_threads,
*diffusion_params.x,
*diffusion_params.timesteps,
*extra->input_ids);
}
};
} // namespace SenseNovaU1
#endif // __SD_MODEL_DIFFUSION_SENSENOVA_U1_H__
+175 -49
View File
@@ -1,6 +1,7 @@
#ifndef __SD_MODEL_DIFFUSION_WAN_HPP__
#define __SD_MODEL_DIFFUSION_WAN_HPP__
#include <algorithm>
#include <cinttypes>
#include <map>
#include <memory>
@@ -19,25 +20,30 @@ namespace WAN {
constexpr int WAN_GRAPH_SIZE = 10240;
struct WanConfig {
std::string model_type = "t2v";
std::tuple<int, int, int> patch_size = {1, 2, 2};
int64_t text_len = 512;
int64_t in_dim = 16;
int64_t dim = 2048;
int64_t ffn_dim = 8192;
int freq_dim = 256;
int64_t text_dim = 4096;
int64_t out_dim = 16;
int64_t num_heads = 16;
int num_layers = 32;
int vace_layers = 0;
int64_t vace_in_dim = 96;
std::map<int, int> vace_layers_mapping = {};
bool qk_norm = true;
bool cross_attn_norm = true;
float eps = 1e-6f;
int64_t flf_pos_embed_token_number = 0;
int theta = 10000;
std::string model_type = "t2v";
std::tuple<int, int, int> patch_size = {1, 2, 2};
int64_t text_len = 512;
int64_t in_dim = 16;
int64_t dim = 2048;
int64_t ffn_dim = 8192;
int freq_dim = 256;
int64_t text_dim = 4096;
int64_t out_dim = 16;
int64_t num_heads = 16;
int num_layers = 32;
int vace_layers = 0;
int64_t vace_in_dim = 96;
std::map<int, int> vace_layers_mapping = {};
int64_t audio_dim = 1024;
int num_audio_token = 4; // excludes the learned padding token
std::vector<int> audio_inject_layers = {};
std::map<int, int> audio_inject_mapping = {}; // block index -> injector index
std::string adain_mode = "attn_norm";
bool qk_norm = true;
bool cross_attn_norm = true;
float eps = 1e-6f;
int64_t flf_pos_embed_token_number = 0;
int theta = 10000;
// wan2.1 1.3B: 1536/12, wan2.1/2.2 14B: 5120/40, wan2.2 5B: 3074/24
std::vector<int> axes_dim = {44, 42, 42};
int64_t axes_dim_sum = 128;
@@ -74,6 +80,10 @@ namespace WAN {
if (name.find("img_emb") != std::string::npos) {
config.model_type = "i2v";
}
if (name.find("audio_injector") != std::string::npos || name.find("casual_audio_encoder") != std::string::npos) {
config.model_type = "s2v";
config.audio_inject_layers = {0, 4, 8, 12, 16, 20, 24, 27, 30, 33, 36, 39};
}
if (name.find("img_emb.emb_pos") != std::string::npos) {
config.flf_pos_embed_token_number = 514;
}
@@ -193,7 +203,7 @@ namespace WAN {
k = norm_k->forward(ctx, k);
auto v = v_proj->forward(ctx, context); // [N, n_context, dim]
x = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, q, k, v, num_heads, nullptr, false, ctx->flash_attn_enabled); // [N, n_token, dim]
x = ggml_ext_attention_ext(ctx, q, k, v, num_heads, nullptr, false, ctx->flash_attn_enabled); // [N, n_token, dim]
x = o_proj->forward(ctx, x); // [N, n_token, dim]
return x;
@@ -255,8 +265,8 @@ namespace WAN {
k_img = norm_k_img->forward(ctx, k_img);
auto v_img = v_img_proj->forward(ctx, context_img); // [N, context_img_len, dim]
auto img_x = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, q, k_img, v_img, num_heads, nullptr, false, ctx->flash_attn_enabled); // [N, n_token, dim]
x = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, q, k, v, num_heads, nullptr, false, ctx->flash_attn_enabled); // [N, n_token, dim]
auto img_x = ggml_ext_attention_ext(ctx, q, k_img, v_img, num_heads, nullptr, false, ctx->flash_attn_enabled); // [N, n_token, dim]
x = ggml_ext_attention_ext(ctx, q, k, v, num_heads, nullptr, false, ctx->flash_attn_enabled); // [N, n_token, dim]
x = ggml_add(ctx->ggml_ctx, x, img_x);
@@ -265,6 +275,13 @@ namespace WAN {
}
};
} // namespace WAN
// Audio injection reuses WanT2VCrossAttention defined above.
#include "model/diffusion/wan_audio.hpp"
namespace WAN {
static ggml_tensor* modulate_add(ggml_context* ctx, ggml_tensor* x, ggml_tensor* e) {
// x: [N, n_token, dim]
// e: [N, 1, dim] or [N, T, 1, dim]
@@ -532,6 +549,13 @@ namespace WAN {
protected:
WanConfig config;
void init_params(ggml_context* ctx, const String2TensorStorage& tensor_storage_map = {}, const std::string prefix = "") override {
if (config.model_type == "s2v") {
enum ggml_type wtype = GGML_TYPE_F32; // elementwise add vs F32 activations
params["trainable_cond_mask.weight"] = ggml_new_tensor_2d(ctx, wtype, config.dim, 3);
}
}
public:
Wan() {}
Wan(WanConfig config)
@@ -554,7 +578,7 @@ namespace WAN {
// blocks
for (int i = 0; i < config.num_layers; i++) {
auto block = std::shared_ptr<GGMLBlock>(new WanAttentionBlock(config.model_type == "t2v",
auto block = std::shared_ptr<GGMLBlock>(new WanAttentionBlock(config.model_type != "i2v",
config.dim,
config.ffn_dim,
config.num_heads,
@@ -595,6 +619,14 @@ namespace WAN {
blocks["vace_patch_embedding"] = std::shared_ptr<GGMLBlock>(new Conv3d(config.vace_in_dim, config.dim, config.patch_size, config.patch_size));
}
if (config.model_type == "s2v") {
blocks["casual_audio_encoder"] = std::make_shared<WanCausalAudioEncoder>(config.audio_dim, config.dim, config.num_audio_token);
blocks["audio_injector"] = std::make_shared<WanAudioInjector>(config.dim, config.num_heads, (int)config.audio_inject_layers.size(), config.qk_norm, config.eps);
for (size_t i = 0; i < config.audio_inject_layers.size(); i++) {
config.audio_inject_mapping[config.audio_inject_layers[i]] = (int)i;
}
}
}
ggml_tensor* pad_to_patch_size(GGMLRunnerContext* ctx,
@@ -642,18 +674,24 @@ namespace WAN {
ggml_tensor* timestep,
ggml_tensor* context,
ggml_tensor* pe,
ggml_tensor* clip_fea = nullptr,
ggml_tensor* vace_context = nullptr,
float vace_strength = 1.f,
int64_t N = 1) {
ggml_tensor* clip_fea = nullptr,
ggml_tensor* vace_context = nullptr,
float vace_strength = 1.f,
int64_t N = 1,
ggml_tensor* audio_embed = nullptr,
ggml_tensor* reference_latent = nullptr) {
// x: [N*C, T, H, W], C => in_dim
// vace_context: [N*vace_in_dim, T, H, W]
// timestep: [N,] or [T]
// context: [N, L, text_dim]
// return: [N, t_len*h_len*w_len, out_dim*pt*ph*pw]
// audio_embed: [layers, T*4, audio_dim]
// reference_latent: [N*C, T_ref, H, W]
// return: [N, (t_len [+ t_ref_len]) * h_len*w_len, out_dim*pt*ph*pw]
GGML_ASSERT(N == 1);
int64_t T = x->ne[2];
auto patch_embedding = std::dynamic_pointer_cast<Conv3d>(blocks["patch_embedding"]);
auto text_embedding_0 = std::dynamic_pointer_cast<Linear>(blocks["text_embedding.0"]);
@@ -670,6 +708,40 @@ namespace WAN {
x = ggml_reshape_3d(ctx->ggml_ctx, x, x->ne[0] * x->ne[1] * x->ne[2], x->ne[3] / N, N); // [N, dim, t_len*h_len*w_len]
x = ggml_ext_cont(ctx->ggml_ctx, ggml_ext_torch_permute(ctx->ggml_ctx, x, 1, 0, 2, 3)); // [N, t_len*h_len*w_len, dim]
ggml_tensor* audio_local = nullptr;
ggml_tensor* audio_global = nullptr;
int64_t seq_len = x->ne[1];
int64_t t_ref_len = 0;
if (config.model_type == "s2v") {
if (audio_embed != nullptr) {
GGML_ASSERT(audio_embed->ne[1] == T * 4);
auto audio_encoder = std::dynamic_pointer_cast<WanCausalAudioEncoder>(blocks["casual_audio_encoder"]);
auto audio_emb = audio_encoder->forward(ctx, audio_embed);
audio_local = audio_emb.first;
audio_global = audio_emb.second;
GGML_ASSERT(audio_local->ne[2] == T);
}
// video tokens get cond_mask[0], reference tokens cond_mask[1]
auto cond_mask = params["trainable_cond_mask.weight"];
auto cm0 = ggml_reshape_3d(ctx->ggml_ctx, ggml_ext_slice(ctx->ggml_ctx, cond_mask, 1, 0, 1), config.dim, 1, 1);
x = ggml_add(ctx->ggml_ctx, x, cm0);
if (reference_latent != nullptr) {
t_ref_len = reference_latent->ne[2];
auto ref = patch_embedding->forward(ctx, reference_latent);
ref = ggml_reshape_3d(ctx->ggml_ctx, ref, ref->ne[0] * ref->ne[1] * ref->ne[2], ref->ne[3] / N, N);
ref = ggml_ext_cont(ctx->ggml_ctx, ggml_ext_torch_permute(ctx->ggml_ctx, ref, 1, 0, 2, 3)); // [N, t_ref*h_len*w_len, dim]
auto cm1 = ggml_reshape_3d(ctx->ggml_ctx, ggml_ext_slice(ctx->ggml_ctx, cond_mask, 1, 1, 2), config.dim, 1, 1);
ref = ggml_add(ctx->ggml_ctx, ref, cm1);
x = ggml_concat(ctx->ggml_ctx, x, ref, 1);
// Reference tokens use timestep 0.
GGML_ASSERT(timestep->ne[0] == T);
timestep = ggml_ext_pad(ctx->ggml_ctx, timestep, (int)t_ref_len, 0, 0, 0);
}
}
// time_embedding
auto e = ggml_ext_timestep_embedding(ctx->ggml_ctx, timestep, config.freq_dim);
e = time_embedding_0->forward(ctx, e);
@@ -714,6 +786,11 @@ namespace WAN {
auto x_orig = x;
std::shared_ptr<WanAudioInjector> audio_injector;
if (audio_local != nullptr) {
audio_injector = std::dynamic_pointer_cast<WanAudioInjector>(blocks["audio_injector"]);
}
for (int i = 0; i < config.num_layers; i++) {
auto block = std::dynamic_pointer_cast<WanAttentionBlock>(blocks["blocks." + std::to_string(i)]);
@@ -731,6 +808,13 @@ namespace WAN {
c_skip = ggml_ext_scale(ctx->ggml_ctx, c_skip, vace_strength);
x = ggml_add(ctx->ggml_ctx, x, c_skip);
}
if (audio_injector != nullptr) {
auto inject_iter = config.audio_inject_mapping.find(i);
if (inject_iter != config.audio_inject_mapping.end()) {
x = audio_injector->forward(ctx, x, seq_len, T, inject_iter->second, audio_local, audio_global);
}
}
sd::ggml_graph_cut::mark_graph_cut(x, "wan.blocks." + std::to_string(i), "x");
if (c != nullptr) {
sd::ggml_graph_cut::mark_graph_cut(c, "wan.blocks." + std::to_string(i), "c");
@@ -747,11 +831,13 @@ namespace WAN {
ggml_tensor* timestep,
ggml_tensor* context,
ggml_tensor* pe,
ggml_tensor* clip_fea = nullptr,
ggml_tensor* time_dim_concat = nullptr,
ggml_tensor* vace_context = nullptr,
float vace_strength = 1.f,
int64_t N = 1) {
ggml_tensor* clip_fea = nullptr,
ggml_tensor* time_dim_concat = nullptr,
ggml_tensor* vace_context = nullptr,
float vace_strength = 1.f,
int64_t N = 1,
ggml_tensor* audio_embed = nullptr,
ggml_tensor* reference_latent = nullptr) {
// Forward pass of DiT.
// x: [N*C, T, H, W]
// timestep: [N,]
@@ -779,7 +865,12 @@ namespace WAN {
t_len = ((x->ne[2] + (std::get<0>(config.patch_size) / 2)) / std::get<0>(config.patch_size));
}
auto out = forward_orig(ctx, x, timestep, context, pe, clip_fea, vace_context, vace_strength, N); // [N, t_len*h_len*w_len, pt*ph*pw*C]
auto out = forward_orig(ctx, x, timestep, context, pe, clip_fea, vace_context, vace_strength, N, audio_embed, reference_latent); // [N, (t_len [+t_ref]) *h_len*w_len, pt*ph*pw*C]
if (reference_latent != nullptr) {
// Exclude reference tokens from the generated video.
out = ggml_ext_slice(ctx->ggml_ctx, out, 1, 0, t_len * h_len * w_len);
}
out = unpatchify(ctx->ggml_ctx, out, t_len, h_len, w_len); // [N*C, (T+pad_t) + (T2+pad_t2), H + pad_h, W + pad_w]
@@ -839,7 +930,10 @@ namespace WAN {
config.text_len = 512;
}
} else if (config.num_layers == 40) {
if (config.model_type == "t2v") {
if (version == VERSION_WAN2_2_S2V) {
desc = "Wan2.2-S2V-14B";
config.in_dim = 16;
} else if (config.model_type == "t2v") {
if (version == VERSION_WAN2_2_I2V) {
desc = "Wan2.2-I2V-14B";
config.in_dim = 36;
@@ -891,7 +985,9 @@ namespace WAN {
const sd::Tensor<float>& c_concat_tensor = {},
const sd::Tensor<float>& time_dim_concat_tensor = {},
const sd::Tensor<float>& vace_context_tensor = {},
float vace_strength = 1.f) {
float vace_strength = 1.f,
const sd::Tensor<float>& audio_embed_tensor = {},
const sd::Tensor<float>& ref_latent_tensor = {}) {
ggml_cgraph* gf = new_graph_custom(WAN_GRAPH_SIZE);
ggml_tensor* x = make_input(x_tensor);
@@ -901,16 +997,33 @@ namespace WAN {
ggml_tensor* c_concat = make_optional_input(c_concat_tensor);
ggml_tensor* time_dim_concat = make_optional_input(time_dim_concat_tensor);
ggml_tensor* vace_context = make_optional_input(vace_context_tensor);
ggml_tensor* audio_embed = make_optional_input(audio_embed_tensor);
ggml_tensor* ref_latent = make_optional_input(ref_latent_tensor);
pe_vec = Rope::gen_wan_pe(static_cast<int>(x->ne[2]),
static_cast<int>(x->ne[1]),
static_cast<int>(x->ne[0]),
std::get<0>(config.patch_size),
std::get<1>(config.patch_size),
std::get<2>(config.patch_size),
1,
config.theta,
config.axes_dim);
pe_vec = Rope::gen_wan_pe(static_cast<int>(x->ne[2]),
static_cast<int>(x->ne[1]),
static_cast<int>(x->ne[0]),
std::get<0>(config.patch_size),
std::get<1>(config.patch_size),
std::get<2>(config.patch_size),
1,
config.theta,
config.axes_dim);
if (ref_latent != nullptr) {
// Match S2V's reference-frame temporal offset.
int t_start = std::max(30, static_cast<int>(x->ne[2]) + 9);
auto ref_pe = Rope::gen_wan_pe(static_cast<int>(ref_latent->ne[2]),
static_cast<int>(ref_latent->ne[1]),
static_cast<int>(ref_latent->ne[0]),
std::get<0>(config.patch_size),
std::get<1>(config.patch_size),
std::get<2>(config.patch_size),
1,
config.theta,
config.axes_dim,
t_start);
pe_vec.insert(pe_vec.end(), ref_pe.begin(), ref_pe.end());
}
int pos_len = static_cast<int>(pe_vec.size() / config.axes_dim_sum / 2);
// LOG_VERBOSE("pos_len %d", pos_len);
auto pe = ggml_new_tensor_4d(compute_ctx, GGML_TYPE_F32, 2, 2, config.axes_dim_sum / 2, pos_len);
@@ -933,7 +1046,10 @@ namespace WAN {
clip_fea,
time_dim_concat,
vace_context,
vace_strength);
vace_strength,
1,
audio_embed,
ref_latent);
ggml_build_forward_expand(gf, out);
@@ -948,9 +1064,11 @@ namespace WAN {
const sd::Tensor<float>& c_concat = {},
const sd::Tensor<float>& time_dim_concat = {},
const sd::Tensor<float>& vace_context = {},
float vace_strength = 1.f) {
float vace_strength = 1.f,
const sd::Tensor<float>& audio_embed = {},
const sd::Tensor<float>& ref_latent = {}) {
auto get_graph = [&]() -> ggml_cgraph* {
return build_graph(x, timesteps, context, clip_fea, c_concat, time_dim_concat, vace_context, vace_strength);
return build_graph(x, timesteps, context, clip_fea, c_concat, time_dim_concat, vace_context, vace_strength, audio_embed, ref_latent);
};
return restore_trailing_singleton_dims(GGMLRunner::compute(get_graph, n_threads, false), x.dim());
@@ -961,6 +1079,12 @@ namespace WAN {
GGML_ASSERT(diffusion_params.x != nullptr);
GGML_ASSERT(diffusion_params.timesteps != nullptr);
const auto* extra = diffusion_extra_as<WanDiffusionExtra>(diffusion_params);
static const std::vector<sd::Tensor<float>> no_ref_latents;
const auto& ref_latents = config.model_type == "s2v" && diffusion_params.ref_latents != nullptr
? *diffusion_params.ref_latents
: no_ref_latents;
const sd::Tensor<float> empty_tensor;
const sd::Tensor<float>& ref_latent = ref_latents.empty() ? empty_tensor : ref_latents[0];
return compute(n_threads,
*diffusion_params.x,
*diffusion_params.timesteps,
@@ -969,7 +1093,9 @@ namespace WAN {
tensor_or_empty(diffusion_params.c_concat),
sd::Tensor<float>(),
tensor_or_empty(extra->vace_context),
extra->vace_strength);
extra->vace_strength,
tensor_or_empty(extra->audio_embed),
ref_latent);
}
void test() {
+215
View File
@@ -0,0 +1,215 @@
#ifndef __SD_MODEL_DIFFUSION_WAN_AUDIO_HPP__
#define __SD_MODEL_DIFFUSION_WAN_AUDIO_HPP__
#include <cstdint>
#include <memory>
#include <string>
#include <utility>
#include <vector>
#include "model/common/ggml_block.hpp"
namespace WAN {
class WanCausalConv1d : public UnaryBlock {
private:
int kernel_size_;
public:
WanCausalConv1d(int64_t in_dim,
int64_t out_dim,
int kernel_size = 3,
int stride = 1)
: kernel_size_(kernel_size) {
blocks["conv"] = std::make_shared<Conv1d>(in_dim, out_dim, kernel_size, stride, 0, 1, 1, true, true);
}
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) override {
// Replicate the first sample for causal left padding.
if (kernel_size_ > 1) {
auto first = ggml_ext_slice(ctx->ggml_ctx, x, 0, 0, 1);
for (int i = 0; i < kernel_size_ - 1; i++) {
x = ggml_concat(ctx->ggml_ctx, first, x, 0);
}
}
return std::dynamic_pointer_cast<Conv1d>(blocks["conv"])->forward(ctx, x);
}
};
class WanMotionEncoder : public GGMLBlock {
private:
int64_t hidden_dim_;
int num_token_;
bool need_global_;
void init_params(ggml_context* ctx, const String2TensorStorage& tensor_storage_map = {}, const std::string prefix = "") override {
// The padding token is combined with F32 activations.
params["padding_tokens"] = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, hidden_dim_);
}
ggml_tensor* conv_norm_silu(GGMLRunnerContext* ctx,
ggml_tensor* x,
const std::string& conv_key,
const std::string& norm_key,
bool to_conv_layout) {
x = std::dynamic_pointer_cast<WanCausalConv1d>(blocks[conv_key])->forward(ctx, x);
x = ggml_permute(ctx->ggml_ctx, x, 1, 0, 2, 3);
x = std::dynamic_pointer_cast<LayerNorm>(blocks[norm_key])->forward(ctx, x);
x = ggml_silu(ctx->ggml_ctx, x);
if (to_conv_layout) {
x = ggml_ext_cont(ctx->ggml_ctx, ggml_permute(ctx->ggml_ctx, x, 1, 0, 2, 3));
}
return x;
}
public:
WanMotionEncoder(int64_t in_dim,
int64_t hidden_dim,
int num_token,
bool need_global = true)
: hidden_dim_(hidden_dim), num_token_(num_token), need_global_(need_global) {
blocks["conv1_local"] = std::make_shared<WanCausalConv1d>(in_dim, hidden_dim / 4 * num_token);
if (need_global) {
blocks["conv1_global"] = std::make_shared<WanCausalConv1d>(in_dim, hidden_dim / 4);
}
blocks["norm1"] = std::make_shared<LayerNorm>(hidden_dim / 4, 1e-6f, false);
blocks["conv2"] = std::make_shared<WanCausalConv1d>(hidden_dim / 4, hidden_dim / 2, 3, 2);
blocks["norm2"] = std::make_shared<LayerNorm>(hidden_dim / 2, 1e-6f, false);
blocks["conv3"] = std::make_shared<WanCausalConv1d>(hidden_dim / 2, hidden_dim, 3, 2);
blocks["norm3"] = std::make_shared<LayerNorm>(hidden_dim, 1e-6f, false);
if (need_global) {
blocks["final_linear"] = std::make_shared<Linear>(hidden_dim, hidden_dim);
}
}
std::pair<ggml_tensor*, ggml_tensor*> forward(GGMLRunnerContext* ctx, ggml_tensor* x) {
auto local = std::dynamic_pointer_cast<WanCausalConv1d>(blocks["conv1_local"])->forward(ctx, x);
auto norm1 = std::dynamic_pointer_cast<LayerNorm>(blocks["norm1"]);
std::vector<ggml_tensor*> tokens;
// Each token group is normalized independently over channels.
for (auto& group : ggml_ext_chunk(ctx->ggml_ctx, local, num_token_, 1)) {
ggml_tensor* s = ggml_permute(ctx->ggml_ctx, group, 1, 0, 2, 3);
s = norm1->forward(ctx, s);
s = ggml_silu(ctx->ggml_ctx, s);
s = ggml_ext_cont(ctx->ggml_ctx, ggml_permute(ctx->ggml_ctx, s, 1, 0, 2, 3));
s = conv_norm_silu(ctx, s, "conv2", "norm2", true);
s = conv_norm_silu(ctx, s, "conv3", "norm3", false);
tokens.push_back(ggml_reshape_3d(ctx->ggml_ctx, s, s->ne[0], 1, s->ne[1]));
}
auto padding = ggml_reshape_3d(ctx->ggml_ctx, params["padding_tokens"], hidden_dim_, 1, 1);
padding = ggml_repeat(ctx->ggml_ctx, padding, tokens[0]);
tokens.push_back(padding);
ggml_tensor* local_out = ggml_ext_vec_concat(ctx->ggml_ctx, tokens, 1);
if (!need_global_) {
return {local_out, nullptr};
}
ggml_tensor* g = conv_norm_silu(ctx, x, "conv1_global", "norm1", true);
g = conv_norm_silu(ctx, g, "conv2", "norm2", true);
g = conv_norm_silu(ctx, g, "conv3", "norm3", false);
g = std::dynamic_pointer_cast<Linear>(blocks["final_linear"])->forward(ctx, g);
return {local_out, g};
}
};
class WanCausalAudioEncoder : public GGMLBlock {
private:
int num_layers_;
void init_params(ggml_context* ctx, const String2TensorStorage& tensor_storage_map = {}, const std::string prefix = "") override {
// Preserve the checkpoint shape for loading; layer mixing requires F32.
auto it = tensor_storage_map.find(prefix + "weights");
if (it != tensor_storage_map.end()) {
params["weights"] = ggml_new_tensor(ctx, GGML_TYPE_F32, it->second.n_dims, it->second.ne);
} else {
params["weights"] = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, num_layers_);
}
}
public:
WanCausalAudioEncoder(int64_t audio_dim,
int64_t dim,
int num_token,
int num_layers = 25)
: num_layers_(num_layers) {
blocks["encoder"] = std::make_shared<WanMotionEncoder>(audio_dim, dim, num_token, true);
}
// features: [layers, frames, audio_dim]; outputs: [T, tokens+1, dim] and [T, dim].
std::pair<ggml_tensor*, ggml_tensor*> forward(GGMLRunnerContext* ctx, ggml_tensor* features) {
auto weights = ggml_silu(ctx->ggml_ctx, params["weights"]);
auto x = ggml_mul(ctx->ggml_ctx, features, ggml_reshape_3d(ctx->ggml_ctx, weights, 1, 1, num_layers_));
x = ggml_div(ctx->ggml_ctx, x, ggml_sum(ctx->ggml_ctx, weights));
// Move the layer axis to ggml dimension 0 for reduction.
x = ggml_ext_cont(ctx->ggml_ctx, ggml_ext_torch_permute(ctx->ggml_ctx, x, 2, 0, 1, 3));
x = ggml_sum_rows(ctx->ggml_ctx, x);
x = ggml_reshape_2d(ctx->ggml_ctx, x, x->ne[1], x->ne[2]);
x = ggml_ext_cont(ctx->ggml_ctx, ggml_ext_torch_permute(ctx->ggml_ctx, x, 1, 0, 2, 3));
return std::dynamic_pointer_cast<WanMotionEncoder>(blocks["encoder"])->forward(ctx, x);
}
};
class WanAudioInjector : public GGMLBlock {
private:
int64_t dim_;
public:
WanAudioInjector(int64_t dim,
int64_t num_heads,
int count,
bool qk_norm = true,
float eps = 1e-6f)
: dim_(dim) {
for (int i = 0; i < count; i++) {
blocks["injector." + std::to_string(i)] =
std::make_shared<WanT2VCrossAttention>(dim, num_heads, qk_norm, eps);
blocks["injector_adain_layers." + std::to_string(i) + ".linear"] =
std::make_shared<Linear>(dim, dim * 2);
}
// S2V AdaLayerNorm uses its own epsilon, independent of attention norms.
blocks["adain_norm"] = std::make_shared<LayerNorm>(dim, 1e-5f, false);
}
// Inject into the video prefix; trailing reference tokens pass through unchanged.
ggml_tensor* forward(GGMLRunnerContext* ctx,
ggml_tensor* x,
int64_t seq_len,
int64_t T,
int injector_id,
ggml_tensor* audio_local,
ggml_tensor* audio_global) {
int64_t n_tok = seq_len / T;
int64_t n_token = x->ne[1];
auto adain_linear = std::dynamic_pointer_cast<Linear>(blocks["injector_adain_layers." + std::to_string(injector_id) + ".linear"]);
auto injector = std::dynamic_pointer_cast<WanT2VCrossAttention>(blocks["injector." + std::to_string(injector_id)]);
auto adain_norm = std::dynamic_pointer_cast<LayerNorm>(blocks["adain_norm"]);
auto temb = ggml_silu(ctx->ggml_ctx, audio_global);
temb = adain_linear->forward(ctx, temb);
auto shift = ggml_ext_slice(ctx->ggml_ctx, temb, 0, 0, dim_);
auto scale = ggml_ext_slice(ctx->ggml_ctx, temb, 0, dim_, dim_ * 2);
shift = ggml_reshape_3d(ctx->ggml_ctx, shift, dim_, 1, T);
scale = ggml_reshape_3d(ctx->ggml_ctx, scale, dim_, 1, T);
auto x_vid = ggml_ext_slice(ctx->ggml_ctx, x, 1, 0, seq_len);
auto h = ggml_reshape_3d(ctx->ggml_ctx, x_vid, dim_, n_tok, T);
h = adain_norm->forward(ctx, h);
h = ggml_add(ctx->ggml_ctx, h, ggml_mul(ctx->ggml_ctx, h, scale));
h = ggml_add(ctx->ggml_ctx, h, shift);
auto res = injector->forward(ctx, h, audio_local, 0);
res = ggml_reshape_2d(ctx->ggml_ctx, res, dim_, seq_len);
auto x_head = ggml_add(ctx->ggml_ctx, x_vid, res);
if (seq_len < n_token) {
auto x_tail = ggml_ext_slice(ctx->ggml_ctx, x, 1, seq_len, n_token);
return ggml_concat(ctx->ggml_ctx, x_head, x_tail, 1);
}
return x_head;
}
};
} // namespace WAN
#endif // __SD_MODEL_DIFFUSION_WAN_AUDIO_HPP__
+4
View File
@@ -145,6 +145,10 @@ protected:
params["position_embedding.weight"] = ggml_new_tensor_2d(ctx, position_wtype, embed_dim, num_positions);
}
enum ggml_op param_usage_op(const std::string& name) const override {
return name == "token_embedding.weight" ? GGML_OP_GET_ROWS : GGML_OP_NONE;
}
public:
CLIPEmbeddings(int64_t embed_dim,
int64_t vocab_size = 49408,
+47 -17
View File
@@ -15,6 +15,7 @@
#include <regex>
#include <set>
#include <sstream>
#include <stdexcept>
#include <string>
#include <utility>
#include <vector>
@@ -31,9 +32,9 @@
#include "model_manager.h"
#include "tokenizers/bpe_tokenizer.h"
#include "tokenizers/gemma_tokenizer.h"
#include "tokenizers/gpt_oss_tokenizer.h"
#include "tokenizers/mistral_tokenizer.h"
#include "tokenizers/qwen2_tokenizer.h"
#include "tokenizers/tokenizer_config.h"
namespace LLM {
constexpr int LLM_GRAPH_SIZE = 65536;
@@ -139,7 +140,8 @@ namespace LLM {
static LLMConfig detect_from_weights(const String2TensorStorage& tensor_storage_map,
const std::string& prefix,
LLMArch arch) {
LLMArch arch,
bool& enable_vision) {
LLMConfig config;
config.arch = arch;
if (arch == LLMArch::MISTRAL_SMALL_3_2 || arch == LLMArch::MINISTRAL_3_3B) {
@@ -230,8 +232,9 @@ namespace LLM {
config.num_experts_per_tok = 4;
}
config.num_layers = 0;
int detected_vision_layers = 0;
config.num_layers = 0;
int detected_vision_layers = 0;
bool out_hidden_size_detected = false;
for (const auto& [name, tensor_storage] : tensor_storage_map) {
if (!starts_with(name, prefix)) {
continue;
@@ -277,6 +280,7 @@ namespace LLM {
if (ends_with(name, "visual.merger.linear_fc2.weight") ||
ends_with(name, "visual.merger.mlp.2.weight")) {
config.vision.out_hidden_size = tensor_storage.ne[1];
out_hidden_size_detected = true;
}
continue;
}
@@ -330,6 +334,20 @@ namespace LLM {
config.vocab_size,
config.hidden_size,
config.intermediate_size);
if (enable_vision && !config.have_vision_weight) {
LOG_WARN("no vision weights detected, vision disabled");
enable_vision = false;
}
// The default would reject valid models, so only compare a detected dim.
if (enable_vision && out_hidden_size_detected &&
config.vision.out_hidden_size != config.hidden_size) {
LOG_ERROR("vision projector output size (%" PRId64 ") does not match LLM hidden size (%" PRId64
"), "
"the vision weights (mmproj) likely belong to a different LLM variant, vision disabled",
config.vision.out_hidden_size,
config.hidden_size);
enable_vision = false;
}
return config;
}
};
@@ -1359,7 +1377,7 @@ namespace LLM {
x = ggml_ext_cont(ctx->ggml_ctx, kqv);
x = ggml_reshape_3d(ctx->ggml_ctx, x, head_dim * num_heads, n_token, N);
} else {
x = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, q, k, v, num_heads, attention_mask, true, false); // [N, n_token, hidden_size]
x = ggml_ext_attention_ext(ctx, q, k, v, num_heads, attention_mask, true, ctx->flash_attn_enabled); // [N, n_token, hidden_size]
}
x = out_proj->forward(ctx, x); // [N, n_token, hidden_size]
@@ -1886,12 +1904,8 @@ namespace LLM {
bool enable_vision_ = false,
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr)
: GGMLRunner(backend, weight_manager),
config(LLMConfig::detect_from_weights(tensor_storage_map, prefix, arch)),
config(LLMConfig::detect_from_weights(tensor_storage_map, prefix, arch, enable_vision_)),
enable_vision(enable_vision_) {
if (enable_vision && !config.have_vision_weight) {
LOG_WARN("no vision weights detected, vision disabled");
enable_vision = false;
}
if (enable_vision) {
LOG_VERBOSE("enable llm vision");
if (config.llama_cpp_style) {
@@ -2338,7 +2352,7 @@ namespace LLM {
};
struct LLMEmbedder {
std::shared_ptr<BPETokenizer> tokenizer;
std::shared_ptr<Tokenizer> tokenizer;
LLMRunner model;
LLMEmbedder(LLMArch arch,
@@ -2346,14 +2360,27 @@ namespace LLM {
const String2TensorStorage& tensor_storage_map = {},
const std::string prefix = "",
bool enable_vision = false,
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr)
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr,
const TokenizerConfig& tokenizers = {})
: model(arch, backend, tensor_storage_map, prefix, enable_vision, weight_manager) {
int pad_id = 151643;
if (arch == LLMArch::MISTRAL_SMALL_3_2 || arch == LLMArch::MINISTRAL_3_3B) {
tokenizer = std::make_shared<MistralTokenizer>();
pad_id = 11;
} else if (arch == LLMArch::GPT_OSS_20B) {
tokenizer = std::make_shared<GPTOSSTokenizer>();
} else {
tokenizer = std::make_shared<Qwen2Tokenizer>();
pad_id = 199999;
} else if (arch == LLMArch::GEMMA2_2B) {
pad_id = 0;
}
tokenizer = tokenizers.create(TokenizerConfig::MAIN, model.config.vocab_size, pad_id);
if (!tokenizer) {
if (arch == LLMArch::GPT_OSS_20B || arch == LLMArch::GEMMA2_2B) {
throw std::runtime_error("GPT-OSS and Gemma 2 require an external tokenizer.json in the main tokenizer slot");
}
if (arch == LLMArch::MISTRAL_SMALL_3_2 || arch == LLMArch::MINISTRAL_3_3B) {
tokenizer = std::make_shared<MistralTokenizer>();
} else {
tokenizer = std::make_shared<Qwen2Tokenizer>();
}
}
}
@@ -2389,7 +2416,10 @@ namespace LLM {
for (const auto& item : parsed_attention) {
const std::string& curr_text = item.first;
float curr_weight = item.second;
std::vector<int> curr_tokens = tokenizer->tokenize(curr_text, nullptr);
std::vector<int> curr_tokens;
if (!tokenizer->tokenize(curr_text, curr_tokens, nullptr)) {
return {};
}
tokens.insert(tokens.end(), curr_tokens.begin(), curr_tokens.end());
weights.insert(weights.end(), curr_tokens.size(), curr_weight);
}
+5 -2
View File
@@ -251,7 +251,7 @@ public:
k = ggml_ext_scale(ctx->ggml_ctx, k, ::sqrtf(static_cast<float>(d_head)), true);
x = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, q, k, v, num_heads, mask); // [N, n_token, d_head * n_head]
x = ggml_ext_attention_ext(ctx, q, k, v, num_heads, mask); // [N, n_token, d_head * n_head]
x = out_proj->forward(ctx, x); // [N, n_token, model_dim]
return {x, past_bias};
@@ -567,7 +567,10 @@ struct T5Embedder {
for (const auto& item : parsed_attention) {
const std::string& curr_text = item.first;
float curr_weight = item.second;
std::vector<int> curr_tokens = tokenizer.encode(curr_text);
std::vector<int> curr_tokens;
if (!tokenizer.encode(curr_text, curr_tokens)) {
return {};
}
tokens.insert(tokens.end(), curr_tokens.begin(), curr_tokens.end());
weights.insert(weights.end(), curr_tokens.size(), curr_weight);
}
+1 -1
View File
@@ -142,7 +142,7 @@ public:
v = ggml_reshape_3d(ctx->ggml_ctx, v, c, h * w, n); // [N, h * w, in_channels]
}
h_ = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, q, k, v, 1, nullptr, false, ctx->flash_attn_enabled);
h_ = ggml_ext_attention_ext(ctx, q, k, v, 1, nullptr, false, ctx->flash_attn_enabled);
if (use_linear) {
h_ = proj_out->forward(ctx, h_); // [N, h * w, in_channels]
+1 -1
View File
@@ -193,7 +193,7 @@ namespace Hunyuan {
v = ggml_reshape_3d(ctx->ggml_ctx, v, w * h * t, c, b); // [b, c, t*h*w]
v = ggml_ext_cont(ctx->ggml_ctx, ggml_ext_torch_permute(ctx->ggml_ctx, v, 1, 0, 2, 3)); // [b, t*h*w, c]
x = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, q, k, v, 1, nullptr, false, ctx->flash_attn_enabled); // [b, t*h*w, c]
x = ggml_ext_attention_ext(ctx, q, k, v, 1, nullptr, false, ctx->flash_attn_enabled); // [b, t*h*w, c]
x = ggml_ext_cont(ctx->ggml_ctx, ggml_permute(ctx->ggml_ctx, x, 1, 0, 2, 3)); // [b, c, t*h*w]
x = ggml_reshape_4d(ctx->ggml_ctx, x, w, h, t, c * b); // [b*c, t, h, w]
+1 -1
View File
@@ -253,7 +253,7 @@ namespace MageVAE {
q = to_patches(ctx->ggml_ctx, q);
k = to_patches(ctx->ggml_ctx, k);
v = to_patches(ctx->ggml_ctx, v);
h = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, q, k, v, 1, nullptr, false, ctx->flash_attn_enabled);
h = ggml_ext_attention_ext(ctx, q, k, v, 1, nullptr, false, ctx->flash_attn_enabled);
h = from_patches(ctx->ggml_ctx, h, np, batch, hp, wp);
if (pad_h > 0) {
h = ggml_ext_slice(ctx->ggml_ctx, h, 1, 0, height);
+1 -2
View File
@@ -174,8 +174,7 @@ namespace MiniMaxH3 {
auto mask = ggml_diag_mask_inf(ctx->ggml_ctx,
ggml_ext_zeros(ctx->ggml_ctx, sequence, sequence, 1, 1),
0);
auto attn_out = ggml_ext_attention_ext(ctx->ggml_ctx,
ctx->backend,
auto attn_out = ggml_ext_attention_ext(ctx,
q,
k,
v,
+1 -2
View File
@@ -291,8 +291,7 @@ namespace MiniMaxH3VAE {
k = ggml_rms_norm(ctx->ggml_ctx, k, 1e-5f);
q = apply_partial_rope(ctx->ggml_ctx, q, pe);
k = apply_partial_rope(ctx->ggml_ctx, k, pe);
auto out = ggml_ext_attention_ext(ctx->ggml_ctx,
ctx->backend,
auto out = ggml_ext_attention_ext(ctx,
q,
k,
v,
+1 -1
View File
@@ -564,7 +564,7 @@ public:
int64_t chunk_frames = 5 * decoder->t_upscale;
int64_t pad = (chunk_frames - (num_frames % chunk_frames)) % chunk_frames;
result = ggml_ext_pad_ext(ctx->ggml_ctx, ctx->backend, result, 0, 0, 0, 0, 0, 0, 0, pad, false, false);
result = ggml_ext_pad_ext(ctx->ggml_ctx, ctx->backend, result, 0, 0, 0, 0, 0, 0, 0, static_cast<int>(pad), false, false);
int64_t num_chunks = (num_frames + pad) / chunk_frames;
auto to_trim = decoder->t_upscale - 1;
+2 -2
View File
@@ -162,11 +162,11 @@ public:
int scale_factor = 8;
if (version == VERSION_LTXAV) {
scale_factor = 32;
} else if (version == VERSION_WAN2_2_TI2V || sd_version_is_hunyuan_video(version) || sd_version_is_mage_flow(version) || sd_version_is_minimax_h3(version)) {
} else if (version == VERSION_WAN2_2_TI2V || version == VERSION_QWEN_IMAGE_2_1 || sd_version_is_hunyuan_video(version) || sd_version_is_mage_flow(version) || sd_version_is_minimax_h3(version)) {
scale_factor = 16;
} else if (sd_version_uses_flux2_vae(version)) {
scale_factor = 16;
} else if (version == VERSION_CHROMA_RADIANCE || version == VERSION_HIDREAM_O1 || sd_version_is_minit2i(version)) {
} else if (version == VERSION_CHROMA_RADIANCE || version == VERSION_HIDREAM_O1 || sd_version_is_minit2i(version) || sd_version_is_sensenova_u1(version)) {
scale_factor = 1;
}
return scale_factor;
+66 -22
View File
@@ -26,6 +26,13 @@ namespace WAN {
bool bias;
void init_params(ggml_context* ctx, const String2TensorStorage& tensor_storage_map = {}, const std::string prefix = "") override {
auto weight = tensor_storage_map.find(prefix + "weight");
if (weight != tensor_storage_map.end() && weight->second.ne[2] == 1 &&
weight->second.ne[3] == in_channels * out_channels) {
// Image VAE exports may retain Conv3d weights with a singleton temporal kernel.
std::get<0>(kernel_size) = 1;
std::get<0>(padding) = 0;
}
params["weight"] = ggml_new_tensor_4d(ctx,
GGML_TYPE_F16,
std::get<2>(kernel_size),
@@ -78,7 +85,8 @@ namespace WAN {
return ggml_ext_conv_3d(ctx->ggml_ctx, ctx->backend, x, w, b, in_channels,
std::get<2>(stride), std::get<1>(stride), std::get<0>(stride),
0, 0, 0,
std::get<2>(dilation), std::get<1>(dilation), std::get<0>(dilation));
std::get<2>(dilation), std::get<1>(dilation), std::get<0>(dilation),
false, ctx->conv3d_direct_enabled);
}
};
@@ -139,7 +147,7 @@ namespace WAN {
std::string mode;
public:
Resample(int64_t dim, const std::string& mode, bool wan2_2 = false)
Resample(int64_t dim, const std::string& mode, bool wan2_2 = false, bool is_2D = false)
: dim(dim), mode(mode) {
if (mode == "upsample2d") {
if (wan2_2) {
@@ -153,12 +161,20 @@ namespace WAN {
} else {
blocks["resample.1"] = std::shared_ptr<GGMLBlock>(new Conv2d(dim, dim / 2, {3, 3}, {1, 1}, {1, 1}));
}
blocks["time_conv"] = std::shared_ptr<GGMLBlock>(new CausalConv3d(dim, dim * 2, {3, 1, 1}, {1, 1, 1}, {1, 0, 0}));
if (is_2D) {
blocks["time_conv"] = std::make_shared<Conv2dBut3d>(dim, dim * 2, std::pair<int, int>{1, 1});
} else {
blocks["time_conv"] = std::shared_ptr<GGMLBlock>(new CausalConv3d(dim, dim * 2, {3, 1, 1}, {1, 1, 1}, {1, 0, 0}));
}
} else if (mode == "downsample2d") {
blocks["resample.1"] = std::shared_ptr<GGMLBlock>(new Conv2d(dim, dim, {3, 3}, {2, 2}));
} else if (mode == "downsample3d") {
blocks["resample.1"] = std::shared_ptr<GGMLBlock>(new Conv2d(dim, dim, {3, 3}, {2, 2}));
blocks["time_conv"] = std::shared_ptr<GGMLBlock>(new CausalConv3d(dim, dim, {3, 1, 1}, {2, 1, 1}, {0, 0, 0}));
if (is_2D) {
blocks["time_conv"] = std::make_shared<Conv2dBut3d>(dim, dim, std::pair<int, int>{1, 1});
} else {
blocks["time_conv"] = std::shared_ptr<GGMLBlock>(new CausalConv3d(dim, dim, {3, 1, 1}, {2, 1, 1}, {0, 0, 0}));
}
} else if (mode == "none") {
// nn.Identity()
} else {
@@ -468,7 +484,7 @@ namespace WAN {
}
if (down_flag) {
std::string mode = temperal_downsample ? "downsample3d" : "downsample2d";
blocks["downsamples." + std::to_string(i)] = std::shared_ptr<GGMLBlock>(new Resample(out_dim, mode, true));
blocks["downsamples." + std::to_string(i)] = std::shared_ptr<GGMLBlock>(new Resample(out_dim, mode, true, is_2D));
i++;
}
}
@@ -531,7 +547,7 @@ namespace WAN {
}
if (up_flag) {
std::string mode = temperal_upsample ? "upsample3d" : "upsample2d";
blocks["upsamples." + std::to_string(i)] = std::shared_ptr<GGMLBlock>(new Resample(out_dim, mode, true));
blocks["upsamples." + std::to_string(i)] = std::shared_ptr<GGMLBlock>(new Resample(out_dim, mode, true, is_2D));
i++;
}
}
@@ -615,8 +631,8 @@ namespace WAN {
auto v = qkv_vec[2];
v = ggml_reshape_3d(ctx->ggml_ctx, v, h * w, c, n); // [t, c, h * w]
v = ggml_cont(ctx->ggml_ctx, ggml_ext_torch_permute(ctx->ggml_ctx, v, 1, 0, 2, 3)); // [t, h * w, c]
x = ggml_ext_attention_ext(ctx->ggml_ctx, ctx->backend, q, k, v, 1, nullptr, false, ctx->flash_attn_enabled); // [t, h * w, c]
v = ggml_cont(ctx->ggml_ctx, ggml_ext_torch_permute(ctx->ggml_ctx, v, 1, 0, 2, 3)); // [t, h * w, c]
x = ggml_ext_attention_ext(ctx, q, k, v, 1, nullptr, false, ctx->flash_attn_enabled); // [t, h * w, c]
x = ggml_ext_cont(ctx->ggml_ctx, ggml_permute(ctx->ggml_ctx, x, 1, 0, 2, 3)); // [t, c, h * w]
x = ggml_reshape_4d(ctx->ggml_ctx, x, w, h, c, n); // [t, c, h, w]
@@ -1053,9 +1069,24 @@ namespace WAN {
input_channels = 4;
}
if (version == VERSION_QWEN_IMAGE_2_1) {
wan2_2 = true;
dec_dim = 144;
z_dim = 64;
input_channels = 4;
dim_mult = {1, 2, 4, 8, 8};
}
if (is_2D) {
temperal_upsample = {false, false, false};
temperal_downsample = {false, false, false};
temperal_upsample.assign(dim_mult.size() - 1, false);
temperal_downsample.assign(dim_mult.size() - 1, false);
}
if (version == VERSION_QWEN_IMAGE_2_1) {
// Temporal shortcut factors still affect single-frame channel grouping.
temperal_upsample = {true, true, true, false};
temperal_downsample = {false, true, true, true};
_conv_num = 2 * (2 + static_cast<int>(dim_mult.size()) * (num_res_blocks + 1)) + 3 + (is_2D ? 0 : 2);
_enc_conv_num = 2 * (2 + static_cast<int>(dim_mult.size()) * num_res_blocks) + 3 + (is_2D ? 0 : 2);
}
if (!decode_only) {
@@ -1270,18 +1301,9 @@ namespace WAN {
SDVersion version = VERSION_WAN2,
std::shared_ptr<RunnerWeightManager> weight_manager = nullptr)
: VAE(version, backend, prefix, weight_manager), decode_only(decode_only) {
bool is_2D = false;
for (const auto& [name, tensor_storage] : tensor_storage_map) {
if (ends_with(name, "decoder.conv1.weight")) {
if (tensor_storage.ne[2] > 3) {
is_2D = true;
}
break;
}
}
if (is_2D) {
LOG_VERBOSE("USING 2D VAE");
}
const auto conv_in = tensor_storage_map.find((prefix.empty() ? "" : prefix + ".") + "decoder.conv1.weight");
const bool is_2D = conv_in != tensor_storage_map.end() && conv_in->second.ne[2] > 3;
LOG_VERBOSE("Wan VAE convolution type: %s", is_2D ? "2D" : "3D");
ae = WanVAE(decode_only, version, is_2D);
ae.init(params_ctx, tensor_storage_map, prefix);
}
@@ -1341,6 +1363,28 @@ namespace WAN {
std_tensor.reshape_(stats_shape);
return {std::move(mean_tensor), std::move(std_tensor)};
}
if (version == VERSION_QWEN_IMAGE_2_1 && latents.shape()[channel_dim] == 64) {
stats_shape[static_cast<size_t>(channel_dim)] = 64;
auto mean_tensor = sd::Tensor<float>::from_vector({0.5126f, 0.7721f, -0.0631f, 1.3506f, -0.7855f, -2.1025f, -0.3458f, 1.3722f,
1.8873f, -1.7177f, -0.6510f, 0.2732f, 0.7562f, -0.6163f, -1.0277f, 3.8363f,
2.0210f, 0.0472f, 0.9320f, 2.0087f, 2.4954f, -0.1391f, -1.4249f, 1.8464f,
-0.5236f, 1.2826f, 3.7046f, -1.3035f, 2.7286f, -1.4518f, -1.9036f, -1.9955f,
-0.0342f, -1.0265f, -0.7636f, 3.0555f, 0.0746f, -3.0751f, -0.1076f, 1.7376f,
-1.0914f, -1.9435f, -0.2784f, -1.3680f, 0.4809f, -0.4433f, 0.3764f, 0.5729f,
-2.0595f, 1.0960f, -1.3260f, -2.0211f, -5.0179f, 0.5275f, 4.0162f, 1.8505f,
0.3026f, 1.9373f, 1.4937f, 0.2632f, 0.5547f, -1.7121f, -0.1562f, 0.0304f});
auto std_tensor = sd::Tensor<float>::from_vector({3.2001f, 3.2936f, 3.4321f, 3.0091f, 3.1061f, 4.0379f, 4.0705f, 3.7910f,
3.0785f, 3.6500f, 3.9308f, 3.0904f, 2.8778f, 3.7675f, 3.7320f, 5.0756f,
3.2864f, 4.0397f, 3.1317f, 4.0443f, 2.9249f, 3.9454f, 3.0988f, 4.2489f,
3.4896f, 3.8513f, 3.9323f, 3.4719f, 3.7498f, 4.2830f, 3.5694f, 4.2467f,
3.9037f, 3.2947f, 5.0770f, 3.5075f, 3.2700f, 3.4767f, 2.8063f, 5.1125f,
3.5327f, 4.7833f, 3.1286f, 4.1819f, 3.8527f, 3.8312f, 3.5605f, 4.3875f,
3.9624f, 4.0168f, 3.5643f, 4.0550f, 5.5614f, 4.2963f, 4.4080f, 3.4959f,
3.8747f, 3.7608f, 3.5735f, 3.1490f, 3.7662f, 3.6746f, 3.4563f, 3.8161f});
mean_tensor.reshape_(stats_shape);
std_tensor.reshape_(stats_shape);
return {std::move(mean_tensor), std::move(std_tensor)};
}
GGML_ABORT("unexpected latent channel dimension %lld for version %d",
(long long)latents.shape()[channel_dim],
version);
+3
View File
@@ -10,6 +10,7 @@ enum class ModelComponent {
VAE,
PreviewVAE,
AudioVAE,
AudioEncoder,
ControlNet,
PhotoMaker,
PuLID,
@@ -38,6 +39,8 @@ inline const char* model_component_name(ModelComponent component) {
return "preview VAE";
case ModelComponent::AudioVAE:
return "audio VAE";
case ModelComponent::AudioEncoder:
return "audio encoder";
case ModelComponent::ControlNet:
return "ControlNet";
case ModelComponent::PhotoMaker:
+12
View File
@@ -57,6 +57,18 @@ bool read_gguf_file(const std::string& file_path,
size_t data_offset = gguf_reader.data_offset();
for (const auto& gguf_tensor_info : gguf_reader.tensors()) {
#ifdef SD_USE_UPSTREAM_GGML
if (static_cast<int>(gguf_tensor_info.type) == SD_TYPE_F8_E4M3 ||
static_cast<int>(gguf_tensor_info.type) == SD_TYPE_F8_E5M2) {
set_error(error, "FP8 is not supported by this ggml build (tensor '" + gguf_tensor_info.name + "')");
return false;
}
#endif
if (static_cast<unsigned>(gguf_tensor_info.type) >= GGML_TYPE_COUNT ||
ggml_get_type_traits(gguf_tensor_info.type)->type_size == 0) {
set_error(error, "unsupported GGUF tensor type (tensor '" + gguf_tensor_info.name + "')");
return false;
}
TensorStorage tensor_storage(
gguf_tensor_info.name,
gguf_tensor_info.type,
+30 -2
View File
@@ -86,10 +86,17 @@ static ggml_type safetensors_dtype_to_ggml_type(const std::string& dtype) {
ttype = GGML_TYPE_F32;
} else if (dtype == "F64") {
ttype = GGML_TYPE_F32;
#ifdef SD_USE_UPSTREAM_GGML
} else if (dtype == "F8_E4M3") {
ttype = GGML_TYPE_F16;
} else if (dtype == "F8_E5M2") {
ttype = GGML_TYPE_F16;
#else
} else if (dtype == "F8_E4M3") {
ttype = GGML_TYPE_F8_E4M3;
} else if (dtype == "F8_E5M2") {
ttype = GGML_TYPE_F8_E5M2;
#endif
} else if (dtype == "I32") {
ttype = GGML_TYPE_I32;
} else if (dtype == "I64") {
@@ -230,6 +237,12 @@ bool read_safetensors_file(const std::string& file_path,
if (!read_comfy_quant_config(file, file_path, name, data_start + begin, end - begin, config, error)) {
return false;
}
#ifdef SD_USE_UPSTREAM_GGML
if (config.format == "int8_tensorwise") {
set_error(error, "INT8 tensorwise/convrot is not supported by this ggml build (tensor '" + name + "')");
return false;
}
#endif
const std::string module_name = name.substr(0, name.size() - std::string(".comfy_quant").size());
comfy_quant_configs.emplace(module_name, std::move(config));
}
@@ -247,6 +260,11 @@ bool read_safetensors_file(const std::string& file_path,
std::string dtype = tensor_info["dtype"];
nlohmann::json shape = tensor_info["shape"];
// ComfyUI FP8 activation scales cancel when inference uses F16/F32 activations.
if (ends_with(name, ".scale_input")) {
continue;
}
size_t begin = tensor_info["data_offsets"][0].get<size_t>();
size_t end = tensor_info["data_offsets"][1].get<size_t>();
if (begin > end || end > file_size_ - data_start) {
@@ -357,10 +375,20 @@ bool read_safetensors_file(const std::string& file_path,
bool tensor_size_ok;
if (dtype == "F8_E4M3") {
tensor_storage.is_f8_e4m3 = true;
tensor_size_ok = (tensor_storage.nbytes() == tensor_data_size);
#ifdef SD_USE_UPSTREAM_GGML
// f8 -> f16
tensor_size_ok = (tensor_storage.nbytes() == tensor_data_size * 2);
#else
tensor_size_ok = (tensor_storage.nbytes() == tensor_data_size);
#endif
} else if (dtype == "F8_E5M2") {
tensor_storage.is_f8_e5m2 = true;
tensor_size_ok = (tensor_storage.nbytes() == tensor_data_size);
#ifdef SD_USE_UPSTREAM_GGML
// f8 -> f16
tensor_size_ok = (tensor_storage.nbytes() == tensor_data_size * 2);
#else
tensor_size_ok = (tensor_storage.nbytes() == tensor_data_size);
#endif
} else if (dtype == "F64") {
tensor_storage.is_f64 = true;
// f64 -> f32
+4
View File
@@ -58,6 +58,10 @@ struct TensorStorage {
int64_t nbytes_to_read() const {
if (is_f64 || is_i64) {
return nbytes() * 2;
#ifdef SD_USE_UPSTREAM_GGML
} else if (is_f8_e4m3 || is_f8_e5m2) {
return nbytes() / 2;
#endif
} else {
return nbytes();
}
+91 -72
View File
@@ -35,51 +35,66 @@
/*================================================= Preprocess ==================================================*/
const char* unused_tensors[] = {
"betas",
"alphas_cumprod_prev",
"sqrt_alphas_cumprod",
"sqrt_one_minus_alphas_cumprod",
"log_one_minus_alphas_cumprod",
"sqrt_recip_alphas_cumprod",
"sqrt_recipm1_alphas_cumprod",
"posterior_variance",
"posterior_log_variance_clipped",
"posterior_mean_coef1",
"posterior_mean_coef2",
"cond_stage_model.transformer.text_model.embeddings.position_ids",
"cond_stage_model.1.model.text_model.embeddings.position_ids",
"cond_stage_model.transformer.vision_model.embeddings.position_ids",
"cond_stage_model.model.logit_scale",
"conditioner.embedders.0.transformer.text_model.embeddings.position_ids",
"conditioner.embedders.0.model.logit_scale",
"conditioner.embedders.1.model.logit_scale",
"model.diffusion_model.time_embedding.cond_proj.weight",
"unet.time_embedding.cond_proj.weight",
"model_ema.decay",
"model_ema.num_updates",
"model_ema.diffusion_model",
"embedding_manager",
"denoiser.sigmas",
"text_encoders.t5xxl.transformer.encoder.embed_tokens.weight", // only used during training
"ztsnr", // Found in some SDXL vpred models
"edm_vpred.sigma_min", // Found in CosXL
// TODO: find another way to avoid the "unknown tensor" for these two
// "edm_vpred.sigma_max", // Used to detect CosXL
// "v_pred", // Used to detect SDXL vpred models
"text_encoders.llm.output.weight",
"text_encoders.llm.lm_head.",
};
bool is_unused_tensor(const std::string& name) {
for (size_t i = 0; i < sizeof(unused_tensors) / sizeof(const char*); i++) {
if (starts_with(name, unused_tensors[i])) {
return true;
}
#ifdef SD_USE_UPSTREAM_GGML
uint16_t f8_e4m3_to_f16(uint8_t f8) {
const uint32_t exponent_bias = 7;
if (f8 == 0xff) {
return ggml_fp32_to_fp16(-NAN);
} else if (f8 == 0x7f) {
return ggml_fp32_to_fp16(NAN);
}
return false;
uint32_t sign = f8 & 0x80;
uint32_t exponent = (f8 & 0x78) >> 3;
uint32_t mantissa = f8 & 0x07;
uint32_t result = sign << 24;
if (exponent == 0) {
if (mantissa > 0) {
exponent = 0x7f - exponent_bias;
// yes, 2 times
if ((mantissa & 0x04) == 0) {
mantissa &= 0x03;
mantissa <<= 1;
exponent -= 1;
}
if ((mantissa & 0x04) == 0) {
mantissa &= 0x03;
mantissa <<= 1;
exponent -= 1;
}
result |= (mantissa & 0x03) << 21;
result |= exponent << 23;
}
} else {
result |= mantissa << 20;
exponent += 0x7f - exponent_bias;
result |= exponent << 23;
}
return ggml_fp32_to_fp16(*reinterpret_cast<const float*>(&result));
}
uint16_t f8_e5m2_to_f16(uint8_t fp8) {
return static_cast<uint16_t>(fp8) << 8;
}
void f8_e4m3_to_f16_vec(uint8_t* src, uint16_t* dst, int64_t n) {
// support inplace op
for (int64_t i = n - 1; i >= 0; i--) {
dst[i] = f8_e4m3_to_f16(src[i]);
}
}
void f8_e5m2_to_f16_vec(uint8_t* src, uint16_t* dst, int64_t n) {
// support inplace op
for (int64_t i = n - 1; i >= 0; i--) {
dst[i] = f8_e5m2_to_f16(src[i]);
}
}
#endif
void f64_to_f32_vec(double* src, float* dst, int64_t n) {
// support inplace op
for (int64_t i = 0; i < n; i++) {
@@ -185,6 +200,15 @@ bool ModelLoader::parse_file(const std::string& file_path, const std::string& pr
}
parsed_dependencies_.push_back(stamp);
if (is_directory(file_path)) {
const std::string diffusers_index_path = path_join(file_path, "model_index.json");
const std::string diffusers_unet_path = path_join(file_path, "unet/diffusion_pytorch_model.safetensors");
const bool has_diffusers_layout = file_exists(diffusers_index_path) || file_exists(diffusers_unet_path);
const std::string safetensors_index_path = path_join(file_path, "model.safetensors.index.json");
if (!has_diffusers_layout && file_exists(safetensors_index_path)) {
LOG_INFO("load %s using root safetensors index", file_path.c_str());
return parse_file(safetensors_index_path, prefix);
}
LOG_INFO("load %s using diffusers format", file_path.c_str());
return init_from_diffusers_file(file_path, prefix);
} else if (is_gguf_file(file_path)) {
@@ -274,10 +298,6 @@ bool ModelLoader::init_from_safetensors_file(const std::string& file_path, const
size_t file_index = add_file_path(file_path);
for (auto& tensor_storage : tensor_storages) {
if (is_unused_tensor(tensor_storage.name)) {
continue;
}
if (!starts_with(tensor_storage.name, prefix)) {
tensor_storage.name = prefix + tensor_storage.name;
}
@@ -346,10 +366,6 @@ bool ModelLoader::init_from_torch_legacy_file(const std::string& file_path, cons
size_t file_index = add_file_path(file_path);
for (auto& tensor_storage : tensor_storages) {
if (is_unused_tensor(tensor_storage.name)) {
continue;
}
if (!starts_with(tensor_storage.name, prefix)) {
tensor_storage.name = prefix + tensor_storage.name;
}
@@ -424,6 +440,7 @@ SDVersion ModelLoader::get_sd_version() const {
bool is_flux2 = false;
bool has_single_block_47 = false;
bool is_wan = false;
bool is_s2v = false;
int64_t patch_embedding_channels = 0;
bool has_img_emb = false;
bool has_middle_block_1 = false;
@@ -463,6 +480,12 @@ SDVersion ModelLoader::get_sd_version() const {
if (tensor_storage.name.find("net.img_embedder.proj1.weight") != std::string::npos) {
return VERSION_MINIT2I;
}
if (tensor_storage.name.find("language_model.model.layers.0.self_attn.q_proj_mot_gen.weight") != std::string::npos) {
return VERSION_SENSENOVA_U1_5;
}
if (tensor_storage.name == "model.diffusion_model.txt_in.text_norm.weight") {
return VERSION_QWEN_IMAGE_2_1;
}
if (tensor_storage.name.find("model.diffusion_model.transformer_blocks.0.img_mod.1.weight") != std::string::npos) {
auto img_in = tensor_storage_map.find("model.diffusion_model.img_in.weight");
if (img_in != tensor_storage_map.end() && img_in->second.ne[0] == 128) {
@@ -510,6 +533,11 @@ SDVersion ModelLoader::get_sd_version() const {
if (tensor_storage.name.find("model.diffusion_model.blocks.0.cross_attn.norm_k.weight") != std::string::npos) {
is_wan = true;
}
if (tensor_storage.name.find("casual_audio_encoder.weights") != std::string::npos ||
tensor_storage.name.find("audio_injector.injector.0.q.weight") != std::string::npos) {
// S2V and T2V-14B share patch_embedding shapes.
is_s2v = true;
}
if (tensor_storage.name.find("model.diffusion_model.patch_embedder.weight") != std::string::npos) {
return VERSION_LINGBOT_VIDEO;
}
@@ -573,6 +601,9 @@ SDVersion ModelLoader::get_sd_version() const {
}
if (is_wan) {
LOG_VERBOSE("patch_embedding_channels %d", patch_embedding_channels);
if (is_s2v) {
return VERSION_WAN2_2_S2V;
}
if (patch_embedding_channels == 184320 && !has_img_emb) {
return VERSION_WAN2_2_I2V;
}
@@ -652,10 +683,6 @@ SDVersion ModelLoader::get_sd_version() const {
std::map<ggml_type, uint32_t> ModelLoader::get_wtype_stat() const {
std::map<ggml_type, uint32_t> wtype_stat;
for (auto& [name, tensor_storage] : tensor_storage_map) {
if (is_unused_tensor(tensor_storage.name)) {
continue;
}
auto iter = wtype_stat.find(tensor_storage.type);
if (iter != wtype_stat.end()) {
iter->second++;
@@ -669,10 +696,6 @@ std::map<ggml_type, uint32_t> ModelLoader::get_wtype_stat() const {
std::map<ggml_type, uint32_t> ModelLoader::get_conditioner_wtype_stat() const {
std::map<ggml_type, uint32_t> wtype_stat;
for (auto& [name, tensor_storage] : tensor_storage_map) {
if (is_unused_tensor(tensor_storage.name)) {
continue;
}
if ((tensor_storage.name.find("text_encoders") == std::string::npos &&
tensor_storage.name.find("cond_stage_model") == std::string::npos &&
tensor_storage.name.find("te.text_model.") == std::string::npos &&
@@ -693,10 +716,6 @@ std::map<ggml_type, uint32_t> ModelLoader::get_conditioner_wtype_stat() const {
std::map<ggml_type, uint32_t> ModelLoader::get_diffusion_model_wtype_stat() const {
std::map<ggml_type, uint32_t> wtype_stat;
for (auto& [name, tensor_storage] : tensor_storage_map) {
if (is_unused_tensor(tensor_storage.name)) {
continue;
}
if (tensor_storage.name.find("model.diffusion_model.") == std::string::npos && tensor_storage.name.find("unet.") == std::string::npos) {
continue;
}
@@ -714,10 +733,6 @@ std::map<ggml_type, uint32_t> ModelLoader::get_diffusion_model_wtype_stat() cons
std::map<ggml_type, uint32_t> ModelLoader::get_vae_wtype_stat() const {
std::map<ggml_type, uint32_t> wtype_stat;
for (auto& [name, tensor_storage] : tensor_storage_map) {
if (is_unused_tensor(tensor_storage.name)) {
continue;
}
if (tensor_storage.name.find("vae.") == std::string::npos &&
tensor_storage.name.find("first_stage_model") == std::string::npos) {
continue;
@@ -801,9 +816,6 @@ void ModelLoader::process_model_files(bool enable_mmap, bool writable_mmap) {
std::vector<TensorStorage> processed_tensor_storages;
for (const auto& [name, tensor_storage] : tensor_storage_map) {
if (is_unused_tensor(tensor_storage.name)) {
continue;
}
processed_tensor_storages.push_back(tensor_storage);
}
@@ -909,6 +921,10 @@ std::vector<MmapTensorStore> ModelLoader::mmap_tensors(std::map<std::string, ggm
if (tensor_storage.is_f64 ||
tensor_storage.is_i64 ||
#ifdef SD_USE_UPSTREAM_GGML
tensor_storage.is_f8_e4m3 ||
tensor_storage.is_f8_e5m2 ||
#endif
tensor_storage.type != dst_tensor->type) {
continue;
}
@@ -1198,6 +1214,12 @@ bool ModelLoader::load_tensors(on_new_tensor_cb_t on_new_tensor_cb,
f64_to_f32_vec((double*)read_buf, (float*)target_buf, tensor_storage.nelements());
} else if (tensor_storage.is_i64) {
i64_to_i32_vec((int64_t*)read_buf, (int32_t*)target_buf, tensor_storage.nelements());
#ifdef SD_USE_UPSTREAM_GGML
} else if (tensor_storage.is_f8_e4m3) {
f8_e4m3_to_f16_vec((uint8_t*)read_buf, (uint16_t*)target_buf, tensor_storage.nelements());
} else if (tensor_storage.is_f8_e5m2) {
f8_e5m2_to_f16_vec((uint8_t*)read_buf, (uint16_t*)target_buf, tensor_storage.nelements());
#endif
}
if (tensor_storage.type != dst_tensor->type) {
if (convert_buf == nullptr) {
@@ -1529,9 +1551,6 @@ int64_t ModelLoader::get_params_mem_size(ggml_backend_t backend, ggml_type type)
int64_t mem_size = 0;
std::vector<TensorStorage> processed_tensor_storages;
for (auto [name, tensor_storage] : tensor_storage_map) {
if (is_unused_tensor(tensor_storage.name)) {
continue;
}
if (tensor_should_be_converted(tensor_storage, type)) {
tensor_storage.type = type;
}
-2
View File
@@ -28,8 +28,6 @@ struct MmapTensorStore {
std::shared_ptr<struct ggml_backend_buffer> mmbuffer;
};
bool is_unused_tensor(const std::string& name);
class ModelLoader {
public:
using FileId = uint64_t;

Some files were not shown because too many files have changed in this diff Show More