Commit bbdcad4 for stable-diffusion.cpp
commit bbdcad48dd1edd13a69f0889803e9cf05a431d67
Author: Aleksei Pakhomov <157724188+aapakhomov@users.noreply.github.com>
Date: Tue Oct 6 12:23:21 2026 +0300
feat: add Z-Image L2P support (#2075)
diff --git a/README.md b/README.md
index 33b1ccb..2305974 100644
--- a/README.md
+++ b/README.md
@@ -52,6 +52,7 @@ API and command-line option may change frequently.***
- [PiD](./docs/pid.md)
- [LongCat Image](./docs/longcat_image.md)
- [Z-Image](./docs/z_image.md)
+ - [Z-Image L2P](./docs/z_image_l2p.md)
- [MiniT2I](./docs/minit2i.md)
- [SenseNova U1.5](./docs/sensenova_u1.md)
- [Ovis-Image](./docs/ovis_image.md)
diff --git a/docs/z_image_l2p.md b/docs/z_image_l2p.md
new file mode 100644
index 0000000..6c56b5e
--- /dev/null
+++ b/docs/z_image_l2p.md
@@ -0,0 +1,21 @@
+# Z-Image L2P
+
+[L2P](https://nju-pcalab.github.io/projects/L2P/) (Latent-to-Pixel) transfers Z-Image-Turbo to pixel space. The VAE is replaced by 16x16 patch tokenization on RGB pixels and a small convolutional local decoder that turns the last DiT hidden states back into pixels, so no VAE is needed. The text encoder is the Qwen3-4B used by Z-Image-Turbo.
+
+## Download weights
+
+- Download L2P (1K)
+ - safetensors: https://huggingface.co/zhen-nan/L2P/tree/main
+- Download Qwen3 4b
+ - safetensors: https://huggingface.co/Comfy-Org/z_image_turbo/tree/main/split_files/text_encoders
+ - gguf: https://huggingface.co/unsloth/Qwen3-4B-Instruct-2507-GGUF/tree/main
+
+## Text-to-image
+
+Do not pass `--vae`. Image dimensions must be multiples of 16. The reference pipeline uses 30 steps with a CFG scale of 2.0.
+
+The released checkpoint keeps part of the transformer in F32. Pass `--type bf16` to load it in BF16, as the reference pipeline runs it; this reduces the diffusion model from about 18.6 GB to 11.8 GB.
+
+```bash
+.\bin\Release\sd-cli.exe --diffusion-model ..\models\diffusion_models\model-1k-merge.safetensors --llm ..\models\text_encoders\qwen_3_4b.safetensors -p "an origami pig on fire in the middle of a dark room with a pentagram on the floor" --cfg-scale 2.0 --steps 30 --sampling-method euler -W 1024 -H 1024 --diffusion-fa --type bf16 -v -o z_image_l2p.png
+```
diff --git a/src/conditioning/conditioner.hpp b/src/conditioning/conditioner.hpp
index 7796a40..5fe83c1 100644
--- a/src/conditioning/conditioner.hpp
+++ b/src/conditioning/conditioner.hpp
@@ -1989,7 +1989,7 @@ struct LLMEmbedder : public Conditioner {
sd_version_is_minimax_h3(version) ||
sd_version_is_mage_flow(version)) {
arch = LLM::LLMArch::QWEN3_VL;
- } else if (sd_version_is_z_image(version) || version == VERSION_OVIS_IMAGE || version == VERSION_FLUX2_KLEIN) {
+ } else if (sd_version_is_z_image(version) || sd_version_is_z_image_l2p(version) || version == VERSION_OVIS_IMAGE || version == VERSION_FLUX2_KLEIN) {
arch = LLM::LLMArch::QWEN3;
}
llm = std::make_shared<LLM::LLMRunner>(arch,
@@ -2953,7 +2953,7 @@ struct LLMEmbedder : public Conditioner {
prompt_attn_range.second = static_cast<int>(prompt.size());
prompt += "<|end|><|start|>assistant<|channel|>analysis<|message|>Need to generate one image according to the description.<|end|><|start|>assistant<|channel|>final<|message|>";
- } else if (sd_version_is_z_image(version)) {
+ } else if (sd_version_is_z_image(version) || sd_version_is_z_image_l2p(version)) {
prompt_template_encode_start_idx = 0;
out_layers = {35}; // -2
diff --git a/src/model.h b/src/model.h
index 277d7b9..19a8213 100644
--- a/src/model.h
+++ b/src/model.h
@@ -64,6 +64,7 @@ enum SDVersion {
VERSION_ESRGAN,
VERSION_PIXART,
VERSION_MING_IMAGE,
+ VERSION_Z_IMAGE_L2P,
VERSION_COUNT,
};
@@ -176,6 +177,10 @@ static inline bool sd_version_is_z_image(SDVersion version) {
return false;
}
+static inline bool sd_version_is_z_image_l2p(SDVersion version) {
+ return version == VERSION_Z_IMAGE_L2P;
+}
+
static inline bool sd_version_is_llada_image(SDVersion version) {
if (version == VERSION_LLADA_IMAGE) {
return true;
@@ -315,6 +320,7 @@ static inline bool sd_version_is_dit(SDVersion version) {
version == VERSION_HIDREAM_O1 ||
sd_version_is_anima(version) ||
sd_version_is_z_image(version) ||
+ sd_version_is_z_image_l2p(version) ||
version == VERSION_MING_IMAGE ||
sd_version_is_llada_image(version) ||
sd_version_is_boogu_image(version) ||
diff --git a/src/model/diffusion/z_image_l2p.hpp b/src/model/diffusion/z_image_l2p.hpp
new file mode 100644
index 0000000..0eb0b14
--- /dev/null
+++ b/src/model/diffusion/z_image_l2p.hpp
@@ -0,0 +1,247 @@
+#ifndef __SD_MODEL_DIFFUSION_Z_IMAGE_L2P_HPP__
+#define __SD_MODEL_DIFFUSION_Z_IMAGE_L2P_HPP__
+
+#include <cmath>
+
+#include "z_image.hpp"
+
+// Ref: https://github.com/TencentYoutuResearch/T2I-L2P/blob/main/diffsynth/models/z_image_dit_L2P.py
+
+namespace ZImageL2P {
+ struct ZImageL2PConfig : ZImage::ZImageConfig {
+ static ZImageL2PConfig detect_from_weights(const String2TensorStorage& tensors, const std::string& prefix) {
+ ZImageL2PConfig config;
+ static_cast<ZImage::ZImageConfig&>(config) = ZImage::ZImageConfig::detect_from_weights(tensors, prefix);
+ config.in_channels = 3;
+ config.out_channels = 3;
+ config.patch_size = 16;
+ auto x_embedder = tensors.find(prefix + ".x_embedder.weight");
+ if (x_embedder != tensors.end()) {
+ config.patch_size = static_cast<int>(std::lround(std::sqrt(static_cast<double>(x_embedder->second.ne[0] / config.in_channels))));
+ }
+ // L2P ships split to_q/to_k/to_v, which name conversion maps to qkv.weight plus
+ // .1/.2 parts; the base detection only sees the q part and undercounts kv heads.
+ auto k_part = tensors.find(prefix + ".layers.0.attention.qkv.weight.1");
+ if (k_part != tensors.end()) {
+ config.num_kv_heads = k_part->second.ne[1] / config.head_dim;
+ }
+ LOG_VERBOSE("z_image_l2p: patch_size = %d, in_channels = %" PRId64 ", num_heads = %" PRId64 ", num_kv_heads = %" PRId64,
+ config.patch_size,
+ config.in_channels,
+ config.num_heads,
+ config.num_kv_heads);
+ return config;
+ }
+ };
+
+ // U-Net over the full-resolution noisy image. Its bottleneck sits at the DiT token
+ // grid, so the number of pooling levels is tied to patch_size == 16.
+ class LocalDecoder : public GGMLBlock {
+ static constexpr int LEVELS = 4;
+ static constexpr int64_t CHANNELS[LEVELS] = {64, 128, 256, 512};
+ static constexpr int64_t BOTTLENECK_CHANNELS = 512;
+
+ public:
+ LocalDecoder(int64_t in_channels, int64_t cond_channels) {
+ int64_t prev = in_channels;
+ for (int i = 0; i < LEVELS; ++i) {
+ blocks["enc" + std::to_string(i + 1) + ".0"] = std::make_shared<Conv2d>(prev, CHANNELS[i], std::pair{3, 3}, std::pair{1, 1}, std::pair{1, 1});
+ prev = CHANNELS[i];
+ }
+ blocks["bottleneck.0"] = std::make_shared<Conv2d>(prev + cond_channels, BOTTLENECK_CHANNELS, std::pair{1, 1});
+ prev = BOTTLENECK_CHANNELS;
+ for (int i = LEVELS - 1; i >= 0; --i) {
+ const std::string level = std::to_string(i + 1);
+ const int64_t out = i == 0 ? CHANNELS[0] : CHANNELS[i - 1];
+ blocks["up" + level + ".1"] = std::make_shared<Conv2d>(prev, prev, std::pair{3, 3}, std::pair{1, 1}, std::pair{1, 1});
+ blocks["dec" + level + ".0"] = std::make_shared<Conv2d>(prev + CHANNELS[i], out, std::pair{3, 3}, std::pair{1, 1}, std::pair{1, 1});
+ prev = out;
+ }
+ blocks["out_conv"] = std::make_shared<Conv2d>(prev, in_channels, std::pair{1, 1});
+ }
+
+ ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x, ggml_tensor* cond) {
+ // x: [N, C, H, W]
+ // cond: [N, cond_channels, H / 16, W / 16]
+ // return: [N, C, H, W]
+ auto gctx = ctx->ggml_ctx;
+ auto conv = [&](const std::string& name, ggml_tensor* h) {
+ return std::dynamic_pointer_cast<Conv2d>(blocks[name])->forward(ctx, h);
+ };
+
+ std::vector<ggml_tensor*> skips;
+ auto h = x;
+ for (int i = 0; i < LEVELS; ++i) {
+ h = ggml_silu(gctx, conv("enc" + std::to_string(i + 1) + ".0", h));
+ skips.push_back(h);
+ h = ggml_pool_2d(gctx, h, GGML_OP_POOL_MAX, 2, 2, 2, 2, 0, 0);
+ }
+
+ h = ggml_concat(gctx, h, cond, 2);
+ h = ggml_silu(gctx, conv("bottleneck.0", h));
+
+ for (int i = LEVELS - 1; i >= 0; --i) {
+ const std::string level = std::to_string(i + 1);
+ h = ggml_upscale(gctx, h, 2, GGML_SCALE_MODE_NEAREST);
+ h = conv("up" + level + ".1", h);
+ h = ggml_concat(gctx, h, skips[i], 2);
+ h = ggml_silu(gctx, conv("dec" + level + ".0", h));
+ }
+ return conv("out_conv", h);
+ }
+ };
+
+ class ZImageL2PModel : public GGMLBlock {
+ ZImageL2PConfig config;
+
+ void init_params(ggml_context* ctx, const String2TensorStorage& tensor_storage_map = {}, const std::string prefix = "") override {
+ params["cap_pad_token"] = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, config.hidden_size);
+ params["x_pad_token"] = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, config.hidden_size);
+ }
+
+ public:
+ explicit ZImageL2PModel(const ZImageL2PConfig& config)
+ : config(config) {
+ blocks["x_embedder"] = std::make_shared<Linear>(config.patch_size * config.patch_size * config.in_channels, config.hidden_size);
+ blocks["t_embedder"] = std::make_shared<TimestepEmbedder>(MIN(config.hidden_size, 1024), 256, 256);
+ blocks["cap_embedder.0"] = std::make_shared<RMSNorm>(config.cap_feat_dim, config.norm_eps);
+ blocks["cap_embedder.1"] = std::make_shared<Linear>(config.cap_feat_dim, config.hidden_size);
+ auto add_blocks = [&](const std::string& prefix, int64_t count, bool modulation) {
+ for (int64_t i = 0; i < count; ++i) {
+ blocks[prefix + std::to_string(i)] = std::make_shared<ZImage::JointTransformerBlock>(
+ static_cast<int>(i), config.hidden_size, config.head_dim, config.num_heads,
+ config.num_kv_heads, config.multiple_of, config.ffn_dim_multiplier,
+ config.norm_eps, config.qk_norm, modulation, true, false, 1e-5f);
+ }
+ };
+ add_blocks("noise_refiner.", config.num_refiner_layers, true);
+ add_blocks("context_refiner.", config.num_refiner_layers, false);
+ add_blocks("layers.", config.num_layers, true);
+ blocks["local_decoder"] = std::make_shared<LocalDecoder>(config.in_channels, config.hidden_size);
+ }
+
+ ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x, ggml_tensor* timestep, ggml_tensor* context, ggml_tensor* pe) {
+ // x: [N, C, H, W], H and W are multiples of patch_size
+ // context: [N, L, cap_feat_dim]
+ // return: [N, C, H, W]
+ auto gctx = ctx->ggml_ctx;
+ const int64_t W = x->ne[0];
+ const int64_t H = x->ne[1];
+ const int64_t N = x->ne[3];
+ const int64_t w = W / config.patch_size;
+ const int64_t h = H / config.patch_size;
+ const int64_t n_img = w * h;
+ const int64_t n_txt = context->ne[1];
+
+ auto img = DiT::patchify(gctx, x, config.patch_size, config.patch_size, false);
+ img = std::dynamic_pointer_cast<Linear>(blocks["x_embedder"])->forward(ctx, img);
+ auto txt = std::dynamic_pointer_cast<RMSNorm>(blocks["cap_embedder.0"])->forward(ctx, context);
+ txt = std::dynamic_pointer_cast<Linear>(blocks["cap_embedder.1"])->forward(ctx, txt);
+ auto t = std::dynamic_pointer_cast<TimestepEmbedder>(blocks["t_embedder"])->forward(ctx, timestep);
+ sd::ggml_graph_cut::mark_graph_cut(txt, "z_image_l2p.prelude", "txt");
+ sd::ggml_graph_cut::mark_graph_cut(img, "z_image_l2p.prelude", "img");
+ sd::ggml_graph_cut::mark_graph_cut(t, "z_image_l2p.prelude", "t_emb");
+
+ const int64_t n_txt_pad = Rope::bound_mod(static_cast<int>(n_txt), ZImage::SEQ_MULTI_OF);
+ if (n_txt_pad > 0) {
+ auto pad = params["cap_pad_token"];
+ txt = ggml_concat(gctx, txt, ggml_repeat_4d(gctx, pad, pad->ne[0], n_txt_pad, N, 1), 1);
+ }
+ const int64_t n_img_pad = Rope::bound_mod(static_cast<int>(n_img), ZImage::SEQ_MULTI_OF);
+ if (n_img_pad > 0) {
+ auto pad = params["x_pad_token"];
+ img = ggml_concat(gctx, img, ggml_repeat_4d(gctx, pad, pad->ne[0], n_img_pad, N, 1), 1);
+ }
+ GGML_ASSERT(txt->ne[1] + img->ne[1] == pe->ne[3]);
+
+ auto txt_pe = ggml_ext_slice(gctx, pe, 3, 0, txt->ne[1]);
+ auto img_pe = ggml_ext_slice(gctx, pe, 3, txt->ne[1], pe->ne[3]);
+ for (int64_t i = 0; i < config.num_refiner_layers; ++i) {
+ txt = std::dynamic_pointer_cast<ZImage::JointTransformerBlock>(blocks["context_refiner." + std::to_string(i)])->forward(ctx, txt, txt_pe);
+ sd::ggml_graph_cut::mark_graph_cut(txt, "z_image_l2p.context_refiner." + std::to_string(i), "txt");
+ }
+ for (int64_t i = 0; i < config.num_refiner_layers; ++i) {
+ img = std::dynamic_pointer_cast<ZImage::JointTransformerBlock>(blocks["noise_refiner." + std::to_string(i)])->forward(ctx, img, img_pe, nullptr, t);
+ sd::ggml_graph_cut::mark_graph_cut(img, "z_image_l2p.noise_refiner." + std::to_string(i), "img");
+ }
+
+ auto txt_img = ggml_concat(gctx, txt, img, 1);
+ for (int64_t i = 0; i < config.num_layers; ++i) {
+ txt_img = std::dynamic_pointer_cast<ZImage::JointTransformerBlock>(blocks["layers." + std::to_string(i)])->forward(ctx, txt_img, pe, nullptr, t);
+ sd::ggml_graph_cut::mark_graph_cut(txt_img, "z_image_l2p.layers." + std::to_string(i), "txt_img");
+ }
+
+ // The local decoder consumes the raw last-layer hidden states: there is no final
+ // norm or adaLN modulation in front of it, unlike Z-Image's final_layer.
+ const int64_t img_start = n_txt + n_txt_pad;
+ auto feat = ggml_ext_slice(gctx, txt_img, 1, img_start, img_start + n_img); // [N, h*w, hidden_size]
+ feat = ggml_reshape_4d(gctx, feat, config.hidden_size, w, h, N); // [N, h, w, hidden_size]
+ feat = ggml_cont(gctx, ggml_permute(gctx, feat, 2, 0, 1, 3)); // [N, hidden_size, h, w]
+ auto out = std::dynamic_pointer_cast<LocalDecoder>(blocks["local_decoder"])->forward(ctx, x, feat); // [N, C, H, W]
+ return ggml_ext_scale(gctx, out, -1.f);
+ }
+ };
+
+ struct ZImageL2PRunner : public DiffusionModelRunner {
+ ZImageL2PConfig config;
+ ZImageL2PModel model;
+ std::vector<float> pe_vec;
+
+ ZImageL2PRunner(ggml_backend_t backend,
+ const String2TensorStorage& tensors,
+ const std::string& prefix,
+ std::shared_ptr<RunnerWeightManager> weight_manager = nullptr)
+ : DiffusionModelRunner(backend, prefix, weight_manager),
+ config(ZImageL2PConfig::detect_from_weights(tensors, prefix)),
+ model(config) {
+ model.init(params_ctx, tensors, prefix);
+ }
+
+ std::string get_desc() override { return "z_image_l2p"; }
+
+ void get_param_tensors(std::map<std::string, ggml_tensor*>& tensors, const std::string& prefix) override {
+ model.get_param_tensors(tensors, prefix);
+ }
+
+ sd::Tensor<float> compute(int n_threads, const DiffusionParams& inputs) override {
+ GGML_ASSERT(inputs.x != nullptr);
+ GGML_ASSERT(inputs.timesteps != nullptr);
+ if (inputs.ref_latents != nullptr && !inputs.ref_latents->empty()) {
+ LOG_ERROR("Z-Image L2P reference-image conditioning is not supported");
+ return {};
+ }
+ if (inputs.context == nullptr || inputs.context->empty()) {
+ LOG_ERROR("Z-Image L2P requires a text condition");
+ return {};
+ }
+ auto graph = [&]() {
+ auto gf = new_graph_custom(ZImage::Z_IMAGE_GRAPH_SIZE);
+ auto x = make_input(*inputs.x);
+ auto t = make_input(*inputs.timesteps);
+ auto context = make_input(*inputs.context);
+ GGML_ASSERT(x->ne[3] == 1);
+ GGML_ASSERT(x->ne[0] % config.patch_size == 0 && x->ne[1] % config.patch_size == 0);
+ pe_vec = finish_rope_pe(Rope::gen_z_image_pe(static_cast<int>(x->ne[1]),
+ static_cast<int>(x->ne[0]),
+ config.patch_size,
+ static_cast<int>(x->ne[3]),
+ static_cast<int>(context->ne[1]),
+ ZImage::SEQ_MULTI_OF,
+ {},
+ Rope::RefIndexMode::FIXED,
+ config.theta,
+ config.axes_dim));
+ int pos_len = static_cast<int>(pe_vec.size() / config.axes_dim_sum / 2);
+ auto pe = ggml_new_tensor_4d(compute_ctx, GGML_TYPE_F32, 2, 2, config.axes_dim_sum / 2, pos_len);
+ set_backend_tensor_data(pe, pe_vec.data());
+ auto ctx = get_context();
+ auto out = model.forward(&ctx, x, t, context, pe);
+ ggml_build_forward_expand(gf, out);
+ return gf;
+ };
+ return restore_trailing_singleton_dims(GGMLRunner::compute(graph, n_threads, false), inputs.x->dim());
+ }
+ };
+} // namespace ZImageL2P
+
+#endif // __SD_MODEL_DIFFUSION_Z_IMAGE_L2P_HPP__
diff --git a/src/model/vae/vae.hpp b/src/model/vae/vae.hpp
index 157b516..5d6e4ee 100644
--- a/src/model/vae/vae.hpp
+++ b/src/model/vae/vae.hpp
@@ -180,7 +180,7 @@ public:
scale_factor = 16;
} else if (sd_version_uses_flux2_vae(version)) {
scale_factor = 16;
- } else if (version == VERSION_CHROMA_RADIANCE || version == VERSION_HIDREAM_O1 || sd_version_is_minit2i(version) || sd_version_is_sensenova_u1(version)) {
+ } else if (version == VERSION_CHROMA_RADIANCE || version == VERSION_HIDREAM_O1 || sd_version_is_minit2i(version) || sd_version_is_sensenova_u1(version) || sd_version_is_z_image_l2p(version)) {
scale_factor = 1;
}
return scale_factor;
diff --git a/src/model_loader.cpp b/src/model_loader.cpp
index bcab0ac..7428e2d 100644
--- a/src/model_loader.cpp
+++ b/src/model_loader.cpp
@@ -527,6 +527,9 @@ SDVersion ModelLoader::get_sd_version() const {
if (tensor_storage_map.find("text_encoders.llm.connector.layers.0.self_attn.q_proj.weight") != tensor_storage_map.end()) {
return VERSION_MING_IMAGE;
}
+ if (tensor_storage_map.find("model.diffusion_model.local_decoder.out_conv.weight") != tensor_storage_map.end()) {
+ return VERSION_Z_IMAGE_L2P;
+ }
return VERSION_Z_IMAGE;
}
if (tensor_storage.name.find("double_stream_layers.0.img_instruct_attn.processor.img_to_q.weight") != std::string::npos) {
diff --git a/src/name_conversion.cpp b/src/name_conversion.cpp
index 0daeda6..8b286e8 100644
--- a/src/name_conversion.cpp
+++ b/src/name_conversion.cpp
@@ -824,6 +824,7 @@ std::string convert_diffusers_dit_to_original_lumina2(std::string name) {
if (z_image_name_map.empty()) {
z_image_name_map["all_x_embedder.2-1."] = "x_embedder.";
z_image_name_map["all_final_layer.2-1."] = "final_layer.";
+ z_image_name_map["all_x_embedder.16-1."] = "x_embedder.";
// --- transformer blocks ---
auto add_attention_map = [&](const std::string& prefix, int num) {
@@ -1036,7 +1037,7 @@ std::string convert_diffusion_model_name(std::string name, std::string prefix, S
name = convert_hunyuan_video_to_original_flux(name);
} else if (version == VERSION_MING_IMAGE) {
name = convert_ming_image_dit_name(name);
- } else if (sd_version_is_z_image(version)) {
+ } else if (sd_version_is_z_image(version) || sd_version_is_z_image_l2p(version)) {
name = convert_diffusers_dit_to_original_lumina2(name);
} else if (sd_version_is_llada_image(version)) {
name = convert_diffusers_dit_to_original_llada_image(name);
diff --git a/src/pipeline/diffusion_engine.cpp b/src/pipeline/diffusion_engine.cpp
index 359b27b..4870cca 100644
--- a/src/pipeline/diffusion_engine.cpp
+++ b/src/pipeline/diffusion_engine.cpp
@@ -107,6 +107,7 @@ const char* model_version_to_str[] = {
"ESRGAN",
"PixArt",
"Ming-Image",
+ "Z-Image L2P",
};
static_assert(VERSION_COUNT == sizeof(model_version_to_str) / sizeof(model_version_to_str[0]),
@@ -1376,6 +1377,7 @@ bool StableDiffusionGGML::build_denoiser() {
sd_version_is_anima(version) ||
sd_version_is_ernie_image(version) ||
sd_version_is_z_image(version) ||
+ sd_version_is_z_image_l2p(version) ||
version == VERSION_MING_IMAGE ||
sd_version_is_llada_image(version) ||
sd_version_is_boogu_image(version) ||
@@ -2172,7 +2174,7 @@ std::vector<float> StableDiffusionGGML::prepare_sample_timesteps(float sigma,
if (version == VERSION_HIDREAM_O1) {
return std::vector<float>{1.0f - (t / static_cast<float>(TIMESTEPS))};
}
- if (sd_version_is_z_image(version) || sd_version_is_ideogram4(version) || version == VERSION_MING_IMAGE) {
+ if (sd_version_is_z_image(version) || sd_version_is_z_image_l2p(version) || sd_version_is_ideogram4(version) || version == VERSION_MING_IMAGE) {
return std::vector<float>{1000.f - t};
}
return std::vector<float>{t};
@@ -2762,6 +2764,8 @@ int StableDiffusionGGML::get_diffusion_model_down_factor() {
if (sd_version_is_dit(version)) {
if (sd_version_is_sensenova_u1(version)) {
down_factor = 32;
+ } else if (sd_version_is_z_image_l2p(version)) {
+ down_factor = 16;
} else if (version == VERSION_QWEN_IMAGE_2_1 || version == VERSION_MING_IMAGE || sd_version_is_wan(version) || sd_version_is_lingbot_video(version) || sd_version_is_minimax_h3(version) || sd_version_is_pixart(version)) {
down_factor = 2;
} else {
@@ -2792,6 +2796,8 @@ int StableDiffusionGGML::get_latent_channel() {
latent_channel = 3;
} else if (sd_version_is_sensenova_u1(version)) {
latent_channel = 3;
+ } else if (sd_version_is_z_image_l2p(version)) {
+ latent_channel = 3;
} else if (sd_version_is_pid(version)) {
latent_channel = 3;
} else if (sd_version_is_sefi_image(version)) {
diff --git a/src/pipeline/model_builders.cpp b/src/pipeline/model_builders.cpp
index f392c36..3b64c8b 100644
--- a/src/pipeline/model_builders.cpp
+++ b/src/pipeline/model_builders.cpp
@@ -37,6 +37,7 @@
#include "model/diffusion/unet.hpp"
#include "model/diffusion/wan.hpp"
#include "model/diffusion/z_image.hpp"
+#include "model/diffusion/z_image_l2p.hpp"
#include "model/vae/auto_encoder_kl.hpp"
#include "model/vae/hunyuan_vae.hpp"
#include "model/vae/ltx_audio_vae.hpp"
@@ -393,6 +394,18 @@ namespace sd::model_builders {
"model.diffusion_model",
version,
weight_manager);
+ } else if (sd_version_is_z_image_l2p(version)) {
+ result.conditioner = std::make_shared<LLMEmbedder>(ctx.backends.runtime_backend(SDBackendModule::TE),
+ tensor_storage_map,
+ version,
+ "",
+ false,
+ weight_manager,
+ tokenizers);
+ result.diffusion = std::make_shared<ZImageL2P::ZImageL2PRunner>(ctx.backends.runtime_backend(SDBackendModule::DIFFUSION),
+ tensor_storage_map,
+ "model.diffusion_model",
+ weight_manager);
} else if (sd_version_is_llada_image(version)) {
result.conditioner = std::make_shared<LLaDAImageEmbedder>(ctx.backends.runtime_backend(SDBackendModule::TE),
tensor_storage_map,
@@ -613,7 +626,7 @@ namespace sd::model_builders {
}
};
- if (version == VERSION_CHROMA_RADIANCE || version == VERSION_HIDREAM_O1 || sd_version_is_minit2i(version) || sd_version_is_sensenova_u1(version)) {
+ if (version == VERSION_CHROMA_RADIANCE || version == VERSION_HIDREAM_O1 || sd_version_is_minit2i(version) || sd_version_is_sensenova_u1(version) || sd_version_is_z_image_l2p(version)) {
LOG_INFO("using FakeVAE");
result.vae = std::make_shared<FakeVAE>(version,
ctx.backends.runtime_backend(SDBackendModule::VAE),