Commit 228c707 for stable-diffusion.cpp
commit 228c707fde018221de74674f1c2f480a9d2b228e
Author: leejet <leejet714@gmail.com>
Date: Fri Oct 9 04:32:36 2026 +0800
perf: use ggml rope apply op on supported backends (#2113)
diff --git a/ggml b/ggml
index 89c4413..66c1a28 160000
--- a/ggml
+++ b/ggml
@@ -1 +1 @@
-Subproject commit 89c4413f5da6fb20cc796f16033d37f129be81fd
+Subproject commit 66c1a2876031613788054a394bafb08d2e9f42dc
diff --git a/src/model/common/rope.hpp b/src/model/common/rope.hpp
index f0fd5b2..93e457d 100644
--- a/src/model/common/rope.hpp
+++ b/src/model/common/rope.hpp
@@ -1067,10 +1067,22 @@ namespace Rope {
return result;
}
- __STATIC_INLINE__ ggml_tensor* apply_rope(ggml_context* ctx,
+ __STATIC_INLINE__ ggml_tensor* apply_rope(GGMLRunnerContext* runner_ctx,
ggml_tensor* x,
ggml_tensor* pe,
bool rope_interleaved = true) {
+ auto ctx = runner_ctx->ggml_ctx;
+#ifndef SD_USE_UPSTREAM_GGML
+ auto backend = runner_ctx->backend;
+ if (backend != nullptr && x->type == GGML_TYPE_F32 && pe->type == GGML_TYPE_F32 &&
+ x->ne[0] % 2 == 0 && pe->ne[0] == 2 && pe->ne[1] == 2 &&
+ pe->ne[2] == x->ne[0] / 2 && pe->ne[3] == x->ne[2]) {
+ auto out = ggml_rope_apply(ctx, x, pe, rope_interleaved);
+ if (ggml_backend_supports_op(backend, out)) {
+ return out;
+ }
+ }
+#endif
// x: [N, L, n_head, d_head]
// pe: [L, d_head/2, 2, 2], [[cos, -sin], [sin, cos]]
int64_t d_head = x->ne[0];
@@ -1126,8 +1138,8 @@ namespace Rope {
// return: [N, L, n_head*d_head]
int64_t n_head = q->ne[1];
- q = apply_rope(ctx->ggml_ctx, q, pe, rope_interleaved); // [N*n_head, L, d_head]
- k = apply_rope(ctx->ggml_ctx, k, pe, rope_interleaved); // [N*n_head, L, d_head]
+ q = apply_rope(ctx, q, pe, rope_interleaved); // [N*n_head, L, d_head]
+ k = apply_rope(ctx, k, pe, rope_interleaved); // [N*n_head, L, d_head]
auto x = ggml_ext_attention_ext(ctx, q, k, v, n_head, mask, true, ctx->flash_attn_enabled, kv_scale); // [N, L, n_head*d_head]
return x;
diff --git a/src/model/diffusion/anima.hpp b/src/model/diffusion/anima.hpp
index 9ace46d..a414d96 100644
--- a/src/model/diffusion/anima.hpp
+++ b/src/model/diffusion/anima.hpp
@@ -235,8 +235,8 @@ namespace Anima {
if (pe_k == nullptr) {
pe_k = pe_q;
}
- auto q_rope = Rope::apply_rope(ctx->ggml_ctx, q4, pe_q, false);
- auto k_rope = Rope::apply_rope(ctx->ggml_ctx, k4, pe_k, false);
+ auto q_rope = Rope::apply_rope(ctx, q4, pe_q, false);
+ auto k_rope = Rope::apply_rope(ctx, k4, pe_k, false);
attn_out = ggml_ext_attention_ext(ctx,
q_rope,
k_rope,
diff --git a/src/model/diffusion/ltxv.hpp b/src/model/diffusion/ltxv.hpp
index 31a65af..e10d5e8 100644
--- a/src/model/diffusion/ltxv.hpp
+++ b/src/model/diffusion/ltxv.hpp
@@ -548,21 +548,22 @@ namespace LTXV {
return build_rope_matrix_from_frequencies(freqs, dim);
}
- __STATIC_INLINE__ ggml_tensor* apply_hidden_rope(ggml_context* ctx,
+ __STATIC_INLINE__ ggml_tensor* apply_hidden_rope(GGMLRunnerContext* runner_ctx,
ggml_tensor* x,
ggml_tensor* pe,
int64_t heads,
int64_t dim_head,
bool rope_interleaved) {
+ auto ctx = runner_ctx->ggml_ctx;
GGML_ASSERT(x->ne[0] == heads * dim_head);
auto x4 = ggml_reshape_4d(ctx, x, dim_head, heads, x->ne[1], x->ne[2]);
if (pe != nullptr && pe->ne[3] == x->ne[1] * heads) {
auto x_flat = ggml_reshape_4d(ctx, x4, dim_head, 1, x->ne[1] * heads, x->ne[2]);
- auto out_flat = Rope::apply_rope(ctx, x_flat, pe, rope_interleaved);
+ auto out_flat = Rope::apply_rope(runner_ctx, x_flat, pe, rope_interleaved);
auto out4 = ggml_reshape_4d(ctx, out_flat, dim_head, heads, x->ne[1], x->ne[2]);
return ggml_reshape_3d(ctx, out4, heads * dim_head, x->ne[1], x->ne[2]);
}
- return Rope::apply_rope(ctx, x4, pe, rope_interleaved);
+ return Rope::apply_rope(runner_ctx, x4, pe, rope_interleaved);
}
struct TimestepEmbedder : public GGMLBlock {
@@ -705,8 +706,8 @@ namespace LTXV {
if (k_pe == nullptr) {
k_pe = pe;
}
- q = apply_hidden_rope(ctx->ggml_ctx, q, pe, heads, dim_head, rope_interleaved);
- k = apply_hidden_rope(ctx->ggml_ctx, k, k_pe, heads, dim_head, rope_interleaved);
+ q = apply_hidden_rope(ctx, q, pe, heads, dim_head, rope_interleaved);
+ k = apply_hidden_rope(ctx, k, k_pe, heads, dim_head, rope_interleaved);
}
auto out = ggml_ext_attention_ext(ctx,
diff --git a/src/model/diffusion/minimax_h3.hpp b/src/model/diffusion/minimax_h3.hpp
index ed7c5ac..7a8693e 100644
--- a/src/model/diffusion/minimax_h3.hpp
+++ b/src/model/diffusion/minimax_h3.hpp
@@ -160,12 +160,13 @@ namespace MiniMaxH3 {
return ggml_reshape_3d(ctx, x, x->ne[0], x->ne[1], x->ne[2] * x->ne[3]);
}
- static ggml_tensor* apply_partial_rope(ggml_context* ctx,
+ static ggml_tensor* apply_partial_rope(GGMLRunnerContext* runner_ctx,
ggml_tensor* x,
ggml_tensor* pe) {
+ auto ctx = runner_ctx->ggml_ctx;
int64_t rot_dim = pe->ne[2] * 2;
GGML_ASSERT(rot_dim <= x->ne[0]);
- auto rotated = Rope::apply_rope(ctx,
+ auto rotated = Rope::apply_rope(runner_ctx,
ggml_ext_slice(ctx, x, 0, 0, rot_dim),
pe,
false);
@@ -209,8 +210,8 @@ namespace MiniMaxH3 {
q = q_norm->forward(ctx, q);
k = k_norm->forward(ctx, k);
if (pe != nullptr) {
- q = apply_partial_rope(ctx->ggml_ctx, q, pe);
- k = apply_partial_rope(ctx->ggml_ctx, k, pe);
+ q = apply_partial_rope(ctx, q, pe);
+ k = apply_partial_rope(ctx, k, pe);
} else {
q = attention_layout(ctx->ggml_ctx, q);
k = attention_layout(ctx->ggml_ctx, k);
diff --git a/src/model/diffusion/qwen_image_2_1.hpp b/src/model/diffusion/qwen_image_2_1.hpp
index 09311c2..1d37d6a 100644
--- a/src/model/diffusion/qwen_image_2_1.hpp
+++ b/src/model/diffusion/qwen_image_2_1.hpp
@@ -188,8 +188,8 @@ namespace Qwen {
auto v = project("to_v");
q = std::dynamic_pointer_cast<RMSNorm>(blocks["norm_q"])->forward(ctx, q);
k = std::dynamic_pointer_cast<RMSNorm>(blocks["norm_k"])->forward(ctx, k);
- q = Rope::apply_rope(ctx->ggml_ctx, q, pe);
- k = Rope::apply_rope(ctx->ggml_ctx, k, pe);
+ q = Rope::apply_rope(ctx, q, pe);
+ k = Rope::apply_rope(ctx, k, pe);
if (cache.mode == QwenImage21PrefixCache::Mode::STORE) {
// Preserve query-first attention evaluation while writing each layer's
// prefix before its full-sequence K/V can accumulate across layers.
diff --git a/src/model/vae/minimax_h3_vae.hpp b/src/model/vae/minimax_h3_vae.hpp
index ec6f623..98cf19c 100644
--- a/src/model/vae/minimax_h3_vae.hpp
+++ b/src/model/vae/minimax_h3_vae.hpp
@@ -228,11 +228,12 @@ namespace MiniMaxH3VAE {
return ggml_reshape_3d(ctx, x, x->ne[0], x->ne[1], x->ne[2] * x->ne[3]);
}
- static ggml_tensor* apply_partial_rope(ggml_context* ctx,
+ static ggml_tensor* apply_partial_rope(GGMLRunnerContext* runner_ctx,
ggml_tensor* x,
ggml_tensor* pe) {
+ auto ctx = runner_ctx->ggml_ctx;
int64_t rot_dim = pe->ne[2] * 2;
- auto rotated = Rope::apply_rope(ctx,
+ auto rotated = Rope::apply_rope(runner_ctx,
ggml_ext_slice(ctx, x, 0, 0, rot_dim),
pe,
false);
@@ -289,8 +290,8 @@ namespace MiniMaxH3VAE {
batch_size);
q = ggml_rms_norm(ctx->ggml_ctx, q, 1e-5f);
k = ggml_rms_norm(ctx->ggml_ctx, k, 1e-5f);
- q = apply_partial_rope(ctx->ggml_ctx, q, pe);
- k = apply_partial_rope(ctx->ggml_ctx, k, pe);
+ q = apply_partial_rope(ctx, q, pe);
+ k = apply_partial_rope(ctx, k, pe);
auto out = ggml_ext_attention_ext(ctx,
q,
k,