Commit 2988060 for stable-diffusion.cpp
commit 2988060a4338a5dcd94721c964a8e87003c1ac8b
Author: leejet <leejet714@gmail.com>
Date: Sun Oct 11 00:01:18 2026 +0800
feat: add TAEQI2.1 support for Qwen Image 2.1 (#2123)
diff --git a/docs/qwen_image_2.1.md b/docs/qwen_image_2.1.md
index 56999bc..5da89c5 100644
--- a/docs/qwen_image_2.1.md
+++ b/docs/qwen_image_2.1.md
@@ -16,6 +16,8 @@ Qwen Image 2.1 supports text-to-image generation and image editing, using Qwen3-
Use `qwen_image_2.1_vae_bf16.safetensors` with this model. The earlier Qwen Image and Wan 2.2 VAE weights are not interchangeable with the Qwen Image 2.1 VAE weights.
+For faster encoding, decoding, or previews, see [TAEQI2.1](taesd.md#qwen-image-21-taeqi21).
+
## Examples
Run the following commands from the build directory. Use image dimensions divisible by 32. The resolution-dependent flow schedule is selected automatically.
diff --git a/docs/taesd.md b/docs/taesd.md
index a41c64d..954086d 100644
--- a/docs/taesd.md
+++ b/docs/taesd.md
@@ -37,3 +37,15 @@ sd.cpp also supports [TAEHV](https://github.com/madebyollin/taehv) (#937), which
```
Then simply replace the `--vae xxx.safetensors` with `--tae xxx.safetensors` in the commands. If it still out of VRAM, add `--vae-conv-direct` to your command though might be slower.
+
+### Qwen Image 2.1 (TAEQI2.1)
+
+For Qwen Image 2.1, use [taeqi2_1](https://github.com/madebyollin/taesd), which supports 64-channel latents, 16x spatial scaling, and RGBA images.
+
+Download the official [safetensors weights](https://huggingface.co/madebyollin/taeqi2_1/blob/main/taeqi2_1.safetensors) directly; no conversion is needed:
+
+```bash
+curl -L -o taeqi2_1.safetensors https://huggingface.co/madebyollin/taeqi2_1/resolve/main/taeqi2_1.safetensors
+```
+
+Replace `--vae PATH` with `--taesd taeqi2_1.safetensors` in the [Qwen Image 2.1 examples](qwen_image_2.1.md) to use it for encoding and decoding. For previews only, keep `--vae PATH` and add `--taesd taeqi2_1.safetensors --taesd-preview-only --preview tae`.
diff --git a/src/model/vae/tae.hpp b/src/model/vae/tae.hpp
index 173b40c..a0700ca 100644
--- a/src/model/vae/tae.hpp
+++ b/src/model/vae/tae.hpp
@@ -84,12 +84,17 @@ class TinyEncoder : public UnaryBlock {
int channels = 64;
int z_channels = 4;
int num_blocks = 3;
+ bool f16;
public:
- TinyEncoder(int z_channels = 4, bool use_midblock_gn = false)
- : z_channels(z_channels) {
- int index = 0;
+ TinyEncoder(int z_channels = 4, bool use_midblock_gn = false, bool f16 = false)
+ : z_channels(z_channels), f16(f16) {
+ in_channels = f16 ? 16 : 3;
+ int index = f16 ? 1 : 0;
blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(in_channels, channels, {3, 3}, {1, 1}, {1, 1}));
+ if (f16) {
+ index++; // nn.ReLU()
+ }
blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new TAEBlock(channels, channels));
blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(channels, channels, {3, 3}, {2, 2}, {1, 1}, {1, 1}, false));
@@ -97,12 +102,14 @@ public:
blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new TAEBlock(channels, channels));
}
- blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(channels, channels, {3, 3}, {2, 2}, {1, 1}, {1, 1}, false));
+ blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(channels, f16 ? channels * 2 : channels, {3, 3}, {2, 2}, {1, 1}, {1, 1}, false));
+ channels *= f16 ? 2 : 1;
for (int i = 0; i < num_blocks; i++) {
blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new TAEBlock(channels, channels));
}
- blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(channels, channels, {3, 3}, {2, 2}, {1, 1}, {1, 1}, false));
+ blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(channels, f16 ? channels * 2 : channels, {3, 3}, {2, 2}, {1, 1}, {1, 1}, false));
+ channels *= f16 ? 2 : 1;
for (int i = 0; i < num_blocks; i++) {
blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new TAEBlock(channels, channels, use_midblock_gn));
}
@@ -114,7 +121,11 @@ public:
// x: [n, in_channels, h, w]
// return: [n, z_channels, h/8, w/8]
- for (int i = 0; i < num_blocks * 3 + 6; i++) {
+ for (int i = f16 ? 1 : 0; i < num_blocks * 3 + 6 + (f16 ? 2 : 0); i++) {
+ if (f16 && i == 2) {
+ x = ggml_relu_inplace(ctx->ggml_ctx, x);
+ continue;
+ }
auto block = std::dynamic_pointer_cast<UnaryBlock>(blocks[std::to_string(i)]);
x = block->forward(ctx, x);
@@ -131,9 +142,11 @@ class TinyDecoder : public UnaryBlock {
int num_blocks = 3;
public:
- TinyDecoder(int z_channels = 4, bool use_midblock_gn = false)
+ TinyDecoder(int z_channels = 4, bool use_midblock_gn = false, bool f16 = false)
: z_channels(z_channels) {
- int index = 0;
+ channels = f16 ? 256 : 64;
+ out_channels = f16 ? 16 : 3;
+ int index = 0;
blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(z_channels, channels, {3, 3}, {1, 1}, {1, 1}));
index++; // nn.ReLU()
@@ -142,13 +155,15 @@ public:
blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new TAEBlock(channels, channels, use_midblock_gn));
}
index++; // nn.Upsample()
- blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(channels, channels, {3, 3}, {1, 1}, {1, 1}, {1, 1}, false));
+ blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(channels, f16 ? channels / 2 : channels, {3, 3}, {1, 1}, {1, 1}, {1, 1}, false));
+ channels /= f16 ? 2 : 1;
for (int i = 0; i < num_blocks; i++) {
blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new TAEBlock(channels, channels));
}
index++; // nn.Upsample()
- blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(channels, channels, {3, 3}, {1, 1}, {1, 1}, {1, 1}, false));
+ blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(channels, f16 ? channels / 2 : channels, {3, 3}, {1, 1}, {1, 1}, {1, 1}, false));
+ channels /= f16 ? 2 : 1;
for (int i = 0; i < num_blocks; i++) {
blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new TAEBlock(channels, channels));
@@ -691,6 +706,7 @@ class TAESD : public GGMLBlock {
protected:
bool decode_only;
bool taef2 = false;
+ bool f16 = false;
public:
int z_channels = 4;
@@ -708,10 +724,14 @@ public:
z_channels = 32;
use_midblock_gn = true;
}
- blocks["decoder.layers"] = std::shared_ptr<GGMLBlock>(new TinyDecoder(z_channels, use_midblock_gn));
+ f16 = version == VERSION_QWEN_IMAGE_2_1;
+ if (f16) {
+ z_channels = 64;
+ }
+ blocks["decoder.layers"] = std::shared_ptr<GGMLBlock>(new TinyDecoder(z_channels, use_midblock_gn, f16));
if (!decode_only) {
- blocks["encoder.layers"] = std::shared_ptr<GGMLBlock>(new TinyEncoder(z_channels, use_midblock_gn));
+ blocks["encoder.layers"] = std::shared_ptr<GGMLBlock>(new TinyEncoder(z_channels, use_midblock_gn, f16));
}
}
@@ -720,10 +740,15 @@ public:
if (taef2) {
z = unpatchify(ctx->ggml_ctx, z, 2);
}
- return decoder->forward(ctx, z);
+ auto x = decoder->forward(ctx, z);
+ return f16 ? unpatchify(ctx->ggml_ctx, x, 2) : x;
}
ggml_tensor* encode(GGMLRunnerContext* ctx, ggml_tensor* x) {
+ if (f16) {
+ GGML_ASSERT(x->ne[0] % 2 == 0 && x->ne[1] % 2 == 0);
+ x = patchify(ctx->ggml_ctx, x, 2);
+ }
auto encoder = std::dynamic_pointer_cast<TinyEncoder>(blocks["encoder.layers"]);
auto z = encoder->forward(ctx, x);
if (taef2) {
diff --git a/src/pipeline/model_builders.cpp b/src/pipeline/model_builders.cpp
index 96b5a97..87cdf40 100644
--- a/src/pipeline/model_builders.cpp
+++ b/src/pipeline/model_builders.cpp
@@ -533,7 +533,7 @@ namespace sd::model_builders {
}
auto create_tae = [&](bool decode_only) -> std::shared_ptr<VAE> {
- if (sd_version_uses_wan_vae(version) || sd_version_is_hunyuan_video(version) || sd_version_is_ltxav(version) || sd_version_is_minimax_h3(version)) {
+ if ((sd_version_uses_wan_vae(version) && version != VERSION_QWEN_IMAGE_2_1) || sd_version_is_hunyuan_video(version) || sd_version_is_ltxav(version) || sd_version_is_minimax_h3(version)) {
return std::make_shared<TinyVideoAutoEncoder>(ctx.backends.runtime_backend(SDBackendModule::VAE),
tensor_storage_map,
"decoder",