Commit bbdcad4 for stable-diffusion.cpp

commit bbdcad48dd1edd13a69f0889803e9cf05a431d67
Author: Aleksei Pakhomov <157724188+aapakhomov@users.noreply.github.com>
Date:   Tue Oct 6 12:23:21 2026 +0300

    feat: add Z-Image L2P support (#2075)

diff --git a/README.md b/README.md
index 33b1ccb..2305974 100644
--- a/README.md
+++ b/README.md
@@ -52,6 +52,7 @@ API and command-line option may change frequently.***
     - [PiD](./docs/pid.md)
     - [LongCat Image](./docs/longcat_image.md)
     - [Z-Image](./docs/z_image.md)
+    - [Z-Image L2P](./docs/z_image_l2p.md)
     - [MiniT2I](./docs/minit2i.md)
     - [SenseNova U1.5](./docs/sensenova_u1.md)
     - [Ovis-Image](./docs/ovis_image.md)
diff --git a/docs/z_image_l2p.md b/docs/z_image_l2p.md
new file mode 100644
index 0000000..6c56b5e
--- /dev/null
+++ b/docs/z_image_l2p.md
@@ -0,0 +1,21 @@
+# Z-Image L2P
+
+[L2P](https://nju-pcalab.github.io/projects/L2P/) (Latent-to-Pixel) transfers Z-Image-Turbo to pixel space. The VAE is replaced by 16x16 patch tokenization on RGB pixels and a small convolutional local decoder that turns the last DiT hidden states back into pixels, so no VAE is needed. The text encoder is the Qwen3-4B used by Z-Image-Turbo.
+
+## Download weights
+
+- Download L2P (1K)
+    - safetensors: https://huggingface.co/zhen-nan/L2P/tree/main
+- Download Qwen3 4b
+    - safetensors: https://huggingface.co/Comfy-Org/z_image_turbo/tree/main/split_files/text_encoders
+    - gguf: https://huggingface.co/unsloth/Qwen3-4B-Instruct-2507-GGUF/tree/main
+
+## Text-to-image
+
+Do not pass `--vae`. Image dimensions must be multiples of 16. The reference pipeline uses 30 steps with a CFG scale of 2.0.
+
+The released checkpoint keeps part of the transformer in F32. Pass `--type bf16` to load it in BF16, as the reference pipeline runs it; this reduces the diffusion model from about 18.6 GB to 11.8 GB.
+
+```bash
+.\bin\Release\sd-cli.exe --diffusion-model ..\models\diffusion_models\model-1k-merge.safetensors --llm ..\models\text_encoders\qwen_3_4b.safetensors -p "an origami pig on fire in the middle of a dark room with a pentagram on the floor" --cfg-scale 2.0 --steps 30 --sampling-method euler -W 1024 -H 1024 --diffusion-fa --type bf16 -v -o z_image_l2p.png
+```
diff --git a/src/conditioning/conditioner.hpp b/src/conditioning/conditioner.hpp
index 7796a40..5fe83c1 100644
--- a/src/conditioning/conditioner.hpp
+++ b/src/conditioning/conditioner.hpp
@@ -1989,7 +1989,7 @@ struct LLMEmbedder : public Conditioner {
                    sd_version_is_minimax_h3(version) ||
                    sd_version_is_mage_flow(version)) {
             arch = LLM::LLMArch::QWEN3_VL;
-        } else if (sd_version_is_z_image(version) || version == VERSION_OVIS_IMAGE || version == VERSION_FLUX2_KLEIN) {
+        } else if (sd_version_is_z_image(version) || sd_version_is_z_image_l2p(version) || version == VERSION_OVIS_IMAGE || version == VERSION_FLUX2_KLEIN) {
             arch = LLM::LLMArch::QWEN3;
         }
         llm        = std::make_shared<LLM::LLMRunner>(arch,
@@ -2953,7 +2953,7 @@ struct LLMEmbedder : public Conditioner {
             prompt_attn_range.second = static_cast<int>(prompt.size());

             prompt += "<|end|><|start|>assistant<|channel|>analysis<|message|>Need to generate one image according to the description.<|end|><|start|>assistant<|channel|>final<|message|>";
-        } else if (sd_version_is_z_image(version)) {
+        } else if (sd_version_is_z_image(version) || sd_version_is_z_image_l2p(version)) {
             prompt_template_encode_start_idx = 0;
             out_layers                       = {35};  // -2

diff --git a/src/model.h b/src/model.h
index 277d7b9..19a8213 100644
--- a/src/model.h
+++ b/src/model.h
@@ -64,6 +64,7 @@ enum SDVersion {
     VERSION_ESRGAN,
     VERSION_PIXART,
     VERSION_MING_IMAGE,
+    VERSION_Z_IMAGE_L2P,
     VERSION_COUNT,
 };

@@ -176,6 +177,10 @@ static inline bool sd_version_is_z_image(SDVersion version) {
     return false;
 }

+static inline bool sd_version_is_z_image_l2p(SDVersion version) {
+    return version == VERSION_Z_IMAGE_L2P;
+}
+
 static inline bool sd_version_is_llada_image(SDVersion version) {
     if (version == VERSION_LLADA_IMAGE) {
         return true;
@@ -315,6 +320,7 @@ static inline bool sd_version_is_dit(SDVersion version) {
         version == VERSION_HIDREAM_O1 ||
         sd_version_is_anima(version) ||
         sd_version_is_z_image(version) ||
+        sd_version_is_z_image_l2p(version) ||
         version == VERSION_MING_IMAGE ||
         sd_version_is_llada_image(version) ||
         sd_version_is_boogu_image(version) ||
diff --git a/src/model/diffusion/z_image_l2p.hpp b/src/model/diffusion/z_image_l2p.hpp
new file mode 100644
index 0000000..0eb0b14
--- /dev/null
+++ b/src/model/diffusion/z_image_l2p.hpp
@@ -0,0 +1,247 @@
+#ifndef __SD_MODEL_DIFFUSION_Z_IMAGE_L2P_HPP__
+#define __SD_MODEL_DIFFUSION_Z_IMAGE_L2P_HPP__
+
+#include <cmath>
+
+#include "z_image.hpp"
+
+// Ref: https://github.com/TencentYoutuResearch/T2I-L2P/blob/main/diffsynth/models/z_image_dit_L2P.py
+
+namespace ZImageL2P {
+    struct ZImageL2PConfig : ZImage::ZImageConfig {
+        static ZImageL2PConfig detect_from_weights(const String2TensorStorage& tensors, const std::string& prefix) {
+            ZImageL2PConfig config;
+            static_cast<ZImage::ZImageConfig&>(config) = ZImage::ZImageConfig::detect_from_weights(tensors, prefix);
+            config.in_channels                         = 3;
+            config.out_channels                        = 3;
+            config.patch_size                          = 16;
+            auto x_embedder                            = tensors.find(prefix + ".x_embedder.weight");
+            if (x_embedder != tensors.end()) {
+                config.patch_size = static_cast<int>(std::lround(std::sqrt(static_cast<double>(x_embedder->second.ne[0] / config.in_channels))));
+            }
+            // L2P ships split to_q/to_k/to_v, which name conversion maps to qkv.weight plus
+            // .1/.2 parts; the base detection only sees the q part and undercounts kv heads.
+            auto k_part = tensors.find(prefix + ".layers.0.attention.qkv.weight.1");
+            if (k_part != tensors.end()) {
+                config.num_kv_heads = k_part->second.ne[1] / config.head_dim;
+            }
+            LOG_VERBOSE("z_image_l2p: patch_size = %d, in_channels = %" PRId64 ", num_heads = %" PRId64 ", num_kv_heads = %" PRId64,
+                        config.patch_size,
+                        config.in_channels,
+                        config.num_heads,
+                        config.num_kv_heads);
+            return config;
+        }
+    };
+
+    // U-Net over the full-resolution noisy image. Its bottleneck sits at the DiT token
+    // grid, so the number of pooling levels is tied to patch_size == 16.
+    class LocalDecoder : public GGMLBlock {
+        static constexpr int LEVELS                  = 4;
+        static constexpr int64_t CHANNELS[LEVELS]    = {64, 128, 256, 512};
+        static constexpr int64_t BOTTLENECK_CHANNELS = 512;
+
+    public:
+        LocalDecoder(int64_t in_channels, int64_t cond_channels) {
+            int64_t prev = in_channels;
+            for (int i = 0; i < LEVELS; ++i) {
+                blocks["enc" + std::to_string(i + 1) + ".0"] = std::make_shared<Conv2d>(prev, CHANNELS[i], std::pair{3, 3}, std::pair{1, 1}, std::pair{1, 1});
+                prev                                         = CHANNELS[i];
+            }
+            blocks["bottleneck.0"] = std::make_shared<Conv2d>(prev + cond_channels, BOTTLENECK_CHANNELS, std::pair{1, 1});
+            prev                   = BOTTLENECK_CHANNELS;
+            for (int i = LEVELS - 1; i >= 0; --i) {
+                const std::string level      = std::to_string(i + 1);
+                const int64_t out            = i == 0 ? CHANNELS[0] : CHANNELS[i - 1];
+                blocks["up" + level + ".1"]  = std::make_shared<Conv2d>(prev, prev, std::pair{3, 3}, std::pair{1, 1}, std::pair{1, 1});
+                blocks["dec" + level + ".0"] = std::make_shared<Conv2d>(prev + CHANNELS[i], out, std::pair{3, 3}, std::pair{1, 1}, std::pair{1, 1});
+                prev                         = out;
+            }
+            blocks["out_conv"] = std::make_shared<Conv2d>(prev, in_channels, std::pair{1, 1});
+        }
+
+        ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x, ggml_tensor* cond) {
+            // x: [N, C, H, W]
+            // cond: [N, cond_channels, H / 16, W / 16]
+            // return: [N, C, H, W]
+            auto gctx = ctx->ggml_ctx;
+            auto conv = [&](const std::string& name, ggml_tensor* h) {
+                return std::dynamic_pointer_cast<Conv2d>(blocks[name])->forward(ctx, h);
+            };
+
+            std::vector<ggml_tensor*> skips;
+            auto h = x;
+            for (int i = 0; i < LEVELS; ++i) {
+                h = ggml_silu(gctx, conv("enc" + std::to_string(i + 1) + ".0", h));
+                skips.push_back(h);
+                h = ggml_pool_2d(gctx, h, GGML_OP_POOL_MAX, 2, 2, 2, 2, 0, 0);
+            }
+
+            h = ggml_concat(gctx, h, cond, 2);
+            h = ggml_silu(gctx, conv("bottleneck.0", h));
+
+            for (int i = LEVELS - 1; i >= 0; --i) {
+                const std::string level = std::to_string(i + 1);
+                h                       = ggml_upscale(gctx, h, 2, GGML_SCALE_MODE_NEAREST);
+                h                       = conv("up" + level + ".1", h);
+                h                       = ggml_concat(gctx, h, skips[i], 2);
+                h                       = ggml_silu(gctx, conv("dec" + level + ".0", h));
+            }
+            return conv("out_conv", h);
+        }
+    };
+
+    class ZImageL2PModel : public GGMLBlock {
+        ZImageL2PConfig config;
+
+        void init_params(ggml_context* ctx, const String2TensorStorage& tensor_storage_map = {}, const std::string prefix = "") override {
+            params["cap_pad_token"] = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, config.hidden_size);
+            params["x_pad_token"]   = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, config.hidden_size);
+        }
+
+    public:
+        explicit ZImageL2PModel(const ZImageL2PConfig& config)
+            : config(config) {
+            blocks["x_embedder"]     = std::make_shared<Linear>(config.patch_size * config.patch_size * config.in_channels, config.hidden_size);
+            blocks["t_embedder"]     = std::make_shared<TimestepEmbedder>(MIN(config.hidden_size, 1024), 256, 256);
+            blocks["cap_embedder.0"] = std::make_shared<RMSNorm>(config.cap_feat_dim, config.norm_eps);
+            blocks["cap_embedder.1"] = std::make_shared<Linear>(config.cap_feat_dim, config.hidden_size);
+            auto add_blocks          = [&](const std::string& prefix, int64_t count, bool modulation) {
+                for (int64_t i = 0; i < count; ++i) {
+                    blocks[prefix + std::to_string(i)] = std::make_shared<ZImage::JointTransformerBlock>(
+                        static_cast<int>(i), config.hidden_size, config.head_dim, config.num_heads,
+                        config.num_kv_heads, config.multiple_of, config.ffn_dim_multiplier,
+                        config.norm_eps, config.qk_norm, modulation, true, false, 1e-5f);
+                }
+            };
+            add_blocks("noise_refiner.", config.num_refiner_layers, true);
+            add_blocks("context_refiner.", config.num_refiner_layers, false);
+            add_blocks("layers.", config.num_layers, true);
+            blocks["local_decoder"] = std::make_shared<LocalDecoder>(config.in_channels, config.hidden_size);
+        }
+
+        ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x, ggml_tensor* timestep, ggml_tensor* context, ggml_tensor* pe) {
+            // x: [N, C, H, W], H and W are multiples of patch_size
+            // context: [N, L, cap_feat_dim]
+            // return: [N, C, H, W]
+            auto gctx           = ctx->ggml_ctx;
+            const int64_t W     = x->ne[0];
+            const int64_t H     = x->ne[1];
+            const int64_t N     = x->ne[3];
+            const int64_t w     = W / config.patch_size;
+            const int64_t h     = H / config.patch_size;
+            const int64_t n_img = w * h;
+            const int64_t n_txt = context->ne[1];
+
+            auto img = DiT::patchify(gctx, x, config.patch_size, config.patch_size, false);
+            img      = std::dynamic_pointer_cast<Linear>(blocks["x_embedder"])->forward(ctx, img);
+            auto txt = std::dynamic_pointer_cast<RMSNorm>(blocks["cap_embedder.0"])->forward(ctx, context);
+            txt      = std::dynamic_pointer_cast<Linear>(blocks["cap_embedder.1"])->forward(ctx, txt);
+            auto t   = std::dynamic_pointer_cast<TimestepEmbedder>(blocks["t_embedder"])->forward(ctx, timestep);
+            sd::ggml_graph_cut::mark_graph_cut(txt, "z_image_l2p.prelude", "txt");
+            sd::ggml_graph_cut::mark_graph_cut(img, "z_image_l2p.prelude", "img");
+            sd::ggml_graph_cut::mark_graph_cut(t, "z_image_l2p.prelude", "t_emb");
+
+            const int64_t n_txt_pad = Rope::bound_mod(static_cast<int>(n_txt), ZImage::SEQ_MULTI_OF);
+            if (n_txt_pad > 0) {
+                auto pad = params["cap_pad_token"];
+                txt      = ggml_concat(gctx, txt, ggml_repeat_4d(gctx, pad, pad->ne[0], n_txt_pad, N, 1), 1);
+            }
+            const int64_t n_img_pad = Rope::bound_mod(static_cast<int>(n_img), ZImage::SEQ_MULTI_OF);
+            if (n_img_pad > 0) {
+                auto pad = params["x_pad_token"];
+                img      = ggml_concat(gctx, img, ggml_repeat_4d(gctx, pad, pad->ne[0], n_img_pad, N, 1), 1);
+            }
+            GGML_ASSERT(txt->ne[1] + img->ne[1] == pe->ne[3]);
+
+            auto txt_pe = ggml_ext_slice(gctx, pe, 3, 0, txt->ne[1]);
+            auto img_pe = ggml_ext_slice(gctx, pe, 3, txt->ne[1], pe->ne[3]);
+            for (int64_t i = 0; i < config.num_refiner_layers; ++i) {
+                txt = std::dynamic_pointer_cast<ZImage::JointTransformerBlock>(blocks["context_refiner." + std::to_string(i)])->forward(ctx, txt, txt_pe);
+                sd::ggml_graph_cut::mark_graph_cut(txt, "z_image_l2p.context_refiner." + std::to_string(i), "txt");
+            }
+            for (int64_t i = 0; i < config.num_refiner_layers; ++i) {
+                img = std::dynamic_pointer_cast<ZImage::JointTransformerBlock>(blocks["noise_refiner." + std::to_string(i)])->forward(ctx, img, img_pe, nullptr, t);
+                sd::ggml_graph_cut::mark_graph_cut(img, "z_image_l2p.noise_refiner." + std::to_string(i), "img");
+            }
+
+            auto txt_img = ggml_concat(gctx, txt, img, 1);
+            for (int64_t i = 0; i < config.num_layers; ++i) {
+                txt_img = std::dynamic_pointer_cast<ZImage::JointTransformerBlock>(blocks["layers." + std::to_string(i)])->forward(ctx, txt_img, pe, nullptr, t);
+                sd::ggml_graph_cut::mark_graph_cut(txt_img, "z_image_l2p.layers." + std::to_string(i), "txt_img");
+            }
+
+            // The local decoder consumes the raw last-layer hidden states: there is no final
+            // norm or adaLN modulation in front of it, unlike Z-Image's final_layer.
+            const int64_t img_start = n_txt + n_txt_pad;
+            auto feat               = ggml_ext_slice(gctx, txt_img, 1, img_start, img_start + n_img);                           // [N, h*w, hidden_size]
+            feat                    = ggml_reshape_4d(gctx, feat, config.hidden_size, w, h, N);                                 // [N, h, w, hidden_size]
+            feat                    = ggml_cont(gctx, ggml_permute(gctx, feat, 2, 0, 1, 3));                                    // [N, hidden_size, h, w]
+            auto out                = std::dynamic_pointer_cast<LocalDecoder>(blocks["local_decoder"])->forward(ctx, x, feat);  // [N, C, H, W]
+            return ggml_ext_scale(gctx, out, -1.f);
+        }
+    };
+
+    struct ZImageL2PRunner : public DiffusionModelRunner {
+        ZImageL2PConfig config;
+        ZImageL2PModel model;
+        std::vector<float> pe_vec;
+
+        ZImageL2PRunner(ggml_backend_t backend,
+                        const String2TensorStorage& tensors,
+                        const std::string& prefix,
+                        std::shared_ptr<RunnerWeightManager> weight_manager = nullptr)
+            : DiffusionModelRunner(backend, prefix, weight_manager),
+              config(ZImageL2PConfig::detect_from_weights(tensors, prefix)),
+              model(config) {
+            model.init(params_ctx, tensors, prefix);
+        }
+
+        std::string get_desc() override { return "z_image_l2p"; }
+
+        void get_param_tensors(std::map<std::string, ggml_tensor*>& tensors, const std::string& prefix) override {
+            model.get_param_tensors(tensors, prefix);
+        }
+
+        sd::Tensor<float> compute(int n_threads, const DiffusionParams& inputs) override {
+            GGML_ASSERT(inputs.x != nullptr);
+            GGML_ASSERT(inputs.timesteps != nullptr);
+            if (inputs.ref_latents != nullptr && !inputs.ref_latents->empty()) {
+                LOG_ERROR("Z-Image L2P reference-image conditioning is not supported");
+                return {};
+            }
+            if (inputs.context == nullptr || inputs.context->empty()) {
+                LOG_ERROR("Z-Image L2P requires a text condition");
+                return {};
+            }
+            auto graph = [&]() {
+                auto gf      = new_graph_custom(ZImage::Z_IMAGE_GRAPH_SIZE);
+                auto x       = make_input(*inputs.x);
+                auto t       = make_input(*inputs.timesteps);
+                auto context = make_input(*inputs.context);
+                GGML_ASSERT(x->ne[3] == 1);
+                GGML_ASSERT(x->ne[0] % config.patch_size == 0 && x->ne[1] % config.patch_size == 0);
+                pe_vec      = finish_rope_pe(Rope::gen_z_image_pe(static_cast<int>(x->ne[1]),
+                                                                  static_cast<int>(x->ne[0]),
+                                                                  config.patch_size,
+                                                                  static_cast<int>(x->ne[3]),
+                                                                  static_cast<int>(context->ne[1]),
+                                                                  ZImage::SEQ_MULTI_OF,
+                                                                  {},
+                                                                  Rope::RefIndexMode::FIXED,
+                                                                  config.theta,
+                                                                  config.axes_dim));
+                int pos_len = static_cast<int>(pe_vec.size() / config.axes_dim_sum / 2);
+                auto pe     = ggml_new_tensor_4d(compute_ctx, GGML_TYPE_F32, 2, 2, config.axes_dim_sum / 2, pos_len);
+                set_backend_tensor_data(pe, pe_vec.data());
+                auto ctx = get_context();
+                auto out = model.forward(&ctx, x, t, context, pe);
+                ggml_build_forward_expand(gf, out);
+                return gf;
+            };
+            return restore_trailing_singleton_dims(GGMLRunner::compute(graph, n_threads, false), inputs.x->dim());
+        }
+    };
+}  // namespace ZImageL2P
+
+#endif  // __SD_MODEL_DIFFUSION_Z_IMAGE_L2P_HPP__
diff --git a/src/model/vae/vae.hpp b/src/model/vae/vae.hpp
index 157b516..5d6e4ee 100644
--- a/src/model/vae/vae.hpp
+++ b/src/model/vae/vae.hpp
@@ -180,7 +180,7 @@ public:
             scale_factor = 16;
         } else if (sd_version_uses_flux2_vae(version)) {
             scale_factor = 16;
-        } else if (version == VERSION_CHROMA_RADIANCE || version == VERSION_HIDREAM_O1 || sd_version_is_minit2i(version) || sd_version_is_sensenova_u1(version)) {
+        } else if (version == VERSION_CHROMA_RADIANCE || version == VERSION_HIDREAM_O1 || sd_version_is_minit2i(version) || sd_version_is_sensenova_u1(version) || sd_version_is_z_image_l2p(version)) {
             scale_factor = 1;
         }
         return scale_factor;
diff --git a/src/model_loader.cpp b/src/model_loader.cpp
index bcab0ac..7428e2d 100644
--- a/src/model_loader.cpp
+++ b/src/model_loader.cpp
@@ -527,6 +527,9 @@ SDVersion ModelLoader::get_sd_version() const {
             if (tensor_storage_map.find("text_encoders.llm.connector.layers.0.self_attn.q_proj.weight") != tensor_storage_map.end()) {
                 return VERSION_MING_IMAGE;
             }
+            if (tensor_storage_map.find("model.diffusion_model.local_decoder.out_conv.weight") != tensor_storage_map.end()) {
+                return VERSION_Z_IMAGE_L2P;
+            }
             return VERSION_Z_IMAGE;
         }
         if (tensor_storage.name.find("double_stream_layers.0.img_instruct_attn.processor.img_to_q.weight") != std::string::npos) {
diff --git a/src/name_conversion.cpp b/src/name_conversion.cpp
index 0daeda6..8b286e8 100644
--- a/src/name_conversion.cpp
+++ b/src/name_conversion.cpp
@@ -824,6 +824,7 @@ std::string convert_diffusers_dit_to_original_lumina2(std::string name) {
     if (z_image_name_map.empty()) {
         z_image_name_map["all_x_embedder.2-1."]  = "x_embedder.";
         z_image_name_map["all_final_layer.2-1."] = "final_layer.";
+        z_image_name_map["all_x_embedder.16-1."] = "x_embedder.";

         // --- transformer blocks ---
         auto add_attention_map = [&](const std::string& prefix, int num) {
@@ -1036,7 +1037,7 @@ std::string convert_diffusion_model_name(std::string name, std::string prefix, S
         name = convert_hunyuan_video_to_original_flux(name);
     } else if (version == VERSION_MING_IMAGE) {
         name = convert_ming_image_dit_name(name);
-    } else if (sd_version_is_z_image(version)) {
+    } else if (sd_version_is_z_image(version) || sd_version_is_z_image_l2p(version)) {
         name = convert_diffusers_dit_to_original_lumina2(name);
     } else if (sd_version_is_llada_image(version)) {
         name = convert_diffusers_dit_to_original_llada_image(name);
diff --git a/src/pipeline/diffusion_engine.cpp b/src/pipeline/diffusion_engine.cpp
index 359b27b..4870cca 100644
--- a/src/pipeline/diffusion_engine.cpp
+++ b/src/pipeline/diffusion_engine.cpp
@@ -107,6 +107,7 @@ const char* model_version_to_str[] = {
     "ESRGAN",
     "PixArt",
     "Ming-Image",
+    "Z-Image L2P",
 };

 static_assert(VERSION_COUNT == sizeof(model_version_to_str) / sizeof(model_version_to_str[0]),
@@ -1376,6 +1377,7 @@ bool StableDiffusionGGML::build_denoiser() {
                    sd_version_is_anima(version) ||
                    sd_version_is_ernie_image(version) ||
                    sd_version_is_z_image(version) ||
+                   sd_version_is_z_image_l2p(version) ||
                    version == VERSION_MING_IMAGE ||
                    sd_version_is_llada_image(version) ||
                    sd_version_is_boogu_image(version) ||
@@ -2172,7 +2174,7 @@ std::vector<float> StableDiffusionGGML::prepare_sample_timesteps(float sigma,
     if (version == VERSION_HIDREAM_O1) {
         return std::vector<float>{1.0f - (t / static_cast<float>(TIMESTEPS))};
     }
-    if (sd_version_is_z_image(version) || sd_version_is_ideogram4(version) || version == VERSION_MING_IMAGE) {
+    if (sd_version_is_z_image(version) || sd_version_is_z_image_l2p(version) || sd_version_is_ideogram4(version) || version == VERSION_MING_IMAGE) {
         return std::vector<float>{1000.f - t};
     }
     return std::vector<float>{t};
@@ -2762,6 +2764,8 @@ int StableDiffusionGGML::get_diffusion_model_down_factor() {
     if (sd_version_is_dit(version)) {
         if (sd_version_is_sensenova_u1(version)) {
             down_factor = 32;
+        } else if (sd_version_is_z_image_l2p(version)) {
+            down_factor = 16;
         } else if (version == VERSION_QWEN_IMAGE_2_1 || version == VERSION_MING_IMAGE || sd_version_is_wan(version) || sd_version_is_lingbot_video(version) || sd_version_is_minimax_h3(version) || sd_version_is_pixart(version)) {
             down_factor = 2;
         } else {
@@ -2792,6 +2796,8 @@ int StableDiffusionGGML::get_latent_channel() {
             latent_channel = 3;
         } else if (sd_version_is_sensenova_u1(version)) {
             latent_channel = 3;
+        } else if (sd_version_is_z_image_l2p(version)) {
+            latent_channel = 3;
         } else if (sd_version_is_pid(version)) {
             latent_channel = 3;
         } else if (sd_version_is_sefi_image(version)) {
diff --git a/src/pipeline/model_builders.cpp b/src/pipeline/model_builders.cpp
index f392c36..3b64c8b 100644
--- a/src/pipeline/model_builders.cpp
+++ b/src/pipeline/model_builders.cpp
@@ -37,6 +37,7 @@
 #include "model/diffusion/unet.hpp"
 #include "model/diffusion/wan.hpp"
 #include "model/diffusion/z_image.hpp"
+#include "model/diffusion/z_image_l2p.hpp"
 #include "model/vae/auto_encoder_kl.hpp"
 #include "model/vae/hunyuan_vae.hpp"
 #include "model/vae/ltx_audio_vae.hpp"
@@ -393,6 +394,18 @@ namespace sd::model_builders {
                                                                       "model.diffusion_model",
                                                                       version,
                                                                       weight_manager);
+        } else if (sd_version_is_z_image_l2p(version)) {
+            result.conditioner = std::make_shared<LLMEmbedder>(ctx.backends.runtime_backend(SDBackendModule::TE),
+                                                               tensor_storage_map,
+                                                               version,
+                                                               "",
+                                                               false,
+                                                               weight_manager,
+                                                               tokenizers);
+            result.diffusion   = std::make_shared<ZImageL2P::ZImageL2PRunner>(ctx.backends.runtime_backend(SDBackendModule::DIFFUSION),
+                                                                            tensor_storage_map,
+                                                                            "model.diffusion_model",
+                                                                            weight_manager);
         } else if (sd_version_is_llada_image(version)) {
             result.conditioner = std::make_shared<LLaDAImageEmbedder>(ctx.backends.runtime_backend(SDBackendModule::TE),
                                                                       tensor_storage_map,
@@ -613,7 +626,7 @@ namespace sd::model_builders {
             }
         };

-        if (version == VERSION_CHROMA_RADIANCE || version == VERSION_HIDREAM_O1 || sd_version_is_minit2i(version) || sd_version_is_sensenova_u1(version)) {
+        if (version == VERSION_CHROMA_RADIANCE || version == VERSION_HIDREAM_O1 || sd_version_is_minit2i(version) || sd_version_is_sensenova_u1(version) || sd_version_is_z_image_l2p(version)) {
             LOG_INFO("using FakeVAE");
             result.vae = std::make_shared<FakeVAE>(version,
                                                    ctx.backends.runtime_backend(SDBackendModule::VAE),