Commit 2988060 for stable-diffusion.cpp

commit 2988060a4338a5dcd94721c964a8e87003c1ac8b
Author: leejet <leejet714@gmail.com>
Date:   Sun Oct 11 00:01:18 2026 +0800

    feat: add TAEQI2.1 support for Qwen Image 2.1 (#2123)

diff --git a/docs/qwen_image_2.1.md b/docs/qwen_image_2.1.md
index 56999bc..5da89c5 100644
--- a/docs/qwen_image_2.1.md
+++ b/docs/qwen_image_2.1.md
@@ -16,6 +16,8 @@ Qwen Image 2.1 supports text-to-image generation and image editing, using Qwen3-

 Use `qwen_image_2.1_vae_bf16.safetensors` with this model. The earlier Qwen Image and Wan 2.2 VAE weights are not interchangeable with the Qwen Image 2.1 VAE weights.

+For faster encoding, decoding, or previews, see [TAEQI2.1](taesd.md#qwen-image-21-taeqi21).
+
 ## Examples

 Run the following commands from the build directory. Use image dimensions divisible by 32. The resolution-dependent flow schedule is selected automatically.
diff --git a/docs/taesd.md b/docs/taesd.md
index a41c64d..954086d 100644
--- a/docs/taesd.md
+++ b/docs/taesd.md
@@ -37,3 +37,15 @@ sd.cpp also supports [TAEHV](https://github.com/madebyollin/taehv) (#937), which
   ```

 Then simply replace the `--vae xxx.safetensors` with `--tae xxx.safetensors` in the commands. If it still out of VRAM, add `--vae-conv-direct` to your command though might be slower.
+
+### Qwen Image 2.1 (TAEQI2.1)
+
+For Qwen Image 2.1, use [taeqi2_1](https://github.com/madebyollin/taesd), which supports 64-channel latents, 16x spatial scaling, and RGBA images.
+
+Download the official [safetensors weights](https://huggingface.co/madebyollin/taeqi2_1/blob/main/taeqi2_1.safetensors) directly; no conversion is needed:
+
+```bash
+curl -L -o taeqi2_1.safetensors https://huggingface.co/madebyollin/taeqi2_1/resolve/main/taeqi2_1.safetensors
+```
+
+Replace `--vae PATH` with `--taesd taeqi2_1.safetensors` in the [Qwen Image 2.1 examples](qwen_image_2.1.md) to use it for encoding and decoding. For previews only, keep `--vae PATH` and add `--taesd taeqi2_1.safetensors --taesd-preview-only --preview tae`.
diff --git a/src/model/vae/tae.hpp b/src/model/vae/tae.hpp
index 173b40c..a0700ca 100644
--- a/src/model/vae/tae.hpp
+++ b/src/model/vae/tae.hpp
@@ -84,12 +84,17 @@ class TinyEncoder : public UnaryBlock {
     int channels    = 64;
     int z_channels  = 4;
     int num_blocks  = 3;
+    bool f16;

 public:
-    TinyEncoder(int z_channels = 4, bool use_midblock_gn = false)
-        : z_channels(z_channels) {
-        int index                       = 0;
+    TinyEncoder(int z_channels = 4, bool use_midblock_gn = false, bool f16 = false)
+        : z_channels(z_channels), f16(f16) {
+        in_channels                     = f16 ? 16 : 3;
+        int index                       = f16 ? 1 : 0;
         blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(in_channels, channels, {3, 3}, {1, 1}, {1, 1}));
+        if (f16) {
+            index++;  // nn.ReLU()
+        }
         blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new TAEBlock(channels, channels));

         blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(channels, channels, {3, 3}, {2, 2}, {1, 1}, {1, 1}, false));
@@ -97,12 +102,14 @@ public:
             blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new TAEBlock(channels, channels));
         }

-        blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(channels, channels, {3, 3}, {2, 2}, {1, 1}, {1, 1}, false));
+        blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(channels, f16 ? channels * 2 : channels, {3, 3}, {2, 2}, {1, 1}, {1, 1}, false));
+        channels *= f16 ? 2 : 1;
         for (int i = 0; i < num_blocks; i++) {
             blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new TAEBlock(channels, channels));
         }

-        blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(channels, channels, {3, 3}, {2, 2}, {1, 1}, {1, 1}, false));
+        blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(channels, f16 ? channels * 2 : channels, {3, 3}, {2, 2}, {1, 1}, {1, 1}, false));
+        channels *= f16 ? 2 : 1;
         for (int i = 0; i < num_blocks; i++) {
             blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new TAEBlock(channels, channels, use_midblock_gn));
         }
@@ -114,7 +121,11 @@ public:
         // x: [n, in_channels, h, w]
         // return: [n, z_channels, h/8, w/8]

-        for (int i = 0; i < num_blocks * 3 + 6; i++) {
+        for (int i = f16 ? 1 : 0; i < num_blocks * 3 + 6 + (f16 ? 2 : 0); i++) {
+            if (f16 && i == 2) {
+                x = ggml_relu_inplace(ctx->ggml_ctx, x);
+                continue;
+            }
             auto block = std::dynamic_pointer_cast<UnaryBlock>(blocks[std::to_string(i)]);

             x = block->forward(ctx, x);
@@ -131,9 +142,11 @@ class TinyDecoder : public UnaryBlock {
     int num_blocks   = 3;

 public:
-    TinyDecoder(int z_channels = 4, bool use_midblock_gn = false)
+    TinyDecoder(int z_channels = 4, bool use_midblock_gn = false, bool f16 = false)
         : z_channels(z_channels) {
-        int index = 0;
+        channels     = f16 ? 256 : 64;
+        out_channels = f16 ? 16 : 3;
+        int index    = 0;

         blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(z_channels, channels, {3, 3}, {1, 1}, {1, 1}));
         index++;  // nn.ReLU()
@@ -142,13 +155,15 @@ public:
             blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new TAEBlock(channels, channels, use_midblock_gn));
         }
         index++;  // nn.Upsample()
-        blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(channels, channels, {3, 3}, {1, 1}, {1, 1}, {1, 1}, false));
+        blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(channels, f16 ? channels / 2 : channels, {3, 3}, {1, 1}, {1, 1}, {1, 1}, false));
+        channels /= f16 ? 2 : 1;

         for (int i = 0; i < num_blocks; i++) {
             blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new TAEBlock(channels, channels));
         }
         index++;  // nn.Upsample()
-        blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(channels, channels, {3, 3}, {1, 1}, {1, 1}, {1, 1}, false));
+        blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new Conv2d(channels, f16 ? channels / 2 : channels, {3, 3}, {1, 1}, {1, 1}, {1, 1}, false));
+        channels /= f16 ? 2 : 1;

         for (int i = 0; i < num_blocks; i++) {
             blocks[std::to_string(index++)] = std::shared_ptr<GGMLBlock>(new TAEBlock(channels, channels));
@@ -691,6 +706,7 @@ class TAESD : public GGMLBlock {
 protected:
     bool decode_only;
     bool taef2 = false;
+    bool f16   = false;

 public:
     int z_channels = 4;
@@ -708,10 +724,14 @@ public:
             z_channels      = 32;
             use_midblock_gn = true;
         }
-        blocks["decoder.layers"] = std::shared_ptr<GGMLBlock>(new TinyDecoder(z_channels, use_midblock_gn));
+        f16 = version == VERSION_QWEN_IMAGE_2_1;
+        if (f16) {
+            z_channels = 64;
+        }
+        blocks["decoder.layers"] = std::shared_ptr<GGMLBlock>(new TinyDecoder(z_channels, use_midblock_gn, f16));

         if (!decode_only) {
-            blocks["encoder.layers"] = std::shared_ptr<GGMLBlock>(new TinyEncoder(z_channels, use_midblock_gn));
+            blocks["encoder.layers"] = std::shared_ptr<GGMLBlock>(new TinyEncoder(z_channels, use_midblock_gn, f16));
         }
     }

@@ -720,10 +740,15 @@ public:
         if (taef2) {
             z = unpatchify(ctx->ggml_ctx, z, 2);
         }
-        return decoder->forward(ctx, z);
+        auto x = decoder->forward(ctx, z);
+        return f16 ? unpatchify(ctx->ggml_ctx, x, 2) : x;
     }

     ggml_tensor* encode(GGMLRunnerContext* ctx, ggml_tensor* x) {
+        if (f16) {
+            GGML_ASSERT(x->ne[0] % 2 == 0 && x->ne[1] % 2 == 0);
+            x = patchify(ctx->ggml_ctx, x, 2);
+        }
         auto encoder = std::dynamic_pointer_cast<TinyEncoder>(blocks["encoder.layers"]);
         auto z       = encoder->forward(ctx, x);
         if (taef2) {
diff --git a/src/pipeline/model_builders.cpp b/src/pipeline/model_builders.cpp
index 96b5a97..87cdf40 100644
--- a/src/pipeline/model_builders.cpp
+++ b/src/pipeline/model_builders.cpp
@@ -533,7 +533,7 @@ namespace sd::model_builders {
         }

         auto create_tae = [&](bool decode_only) -> std::shared_ptr<VAE> {
-            if (sd_version_uses_wan_vae(version) || sd_version_is_hunyuan_video(version) || sd_version_is_ltxav(version) || sd_version_is_minimax_h3(version)) {
+            if ((sd_version_uses_wan_vae(version) && version != VERSION_QWEN_IMAGE_2_1) || sd_version_is_hunyuan_video(version) || sd_version_is_ltxav(version) || sd_version_is_minimax_h3(version)) {
                 return std::make_shared<TinyVideoAutoEncoder>(ctx.backends.runtime_backend(SDBackendModule::VAE),
                                                               tensor_storage_map,
                                                               "decoder",