Commit 228c707 for stable-diffusion.cpp

commit 228c707fde018221de74674f1c2f480a9d2b228e
Author: leejet <leejet714@gmail.com>
Date:   Fri Oct 9 04:32:36 2026 +0800

    perf: use ggml rope apply op on supported backends (#2113)

diff --git a/ggml b/ggml
index 89c4413..66c1a28 160000
--- a/ggml
+++ b/ggml
@@ -1 +1 @@
-Subproject commit 89c4413f5da6fb20cc796f16033d37f129be81fd
+Subproject commit 66c1a2876031613788054a394bafb08d2e9f42dc
diff --git a/src/model/common/rope.hpp b/src/model/common/rope.hpp
index f0fd5b2..93e457d 100644
--- a/src/model/common/rope.hpp
+++ b/src/model/common/rope.hpp
@@ -1067,10 +1067,22 @@ namespace Rope {
         return result;
     }

-    __STATIC_INLINE__ ggml_tensor* apply_rope(ggml_context* ctx,
+    __STATIC_INLINE__ ggml_tensor* apply_rope(GGMLRunnerContext* runner_ctx,
                                               ggml_tensor* x,
                                               ggml_tensor* pe,
                                               bool rope_interleaved = true) {
+        auto ctx = runner_ctx->ggml_ctx;
+#ifndef SD_USE_UPSTREAM_GGML
+        auto backend = runner_ctx->backend;
+        if (backend != nullptr && x->type == GGML_TYPE_F32 && pe->type == GGML_TYPE_F32 &&
+            x->ne[0] % 2 == 0 && pe->ne[0] == 2 && pe->ne[1] == 2 &&
+            pe->ne[2] == x->ne[0] / 2 && pe->ne[3] == x->ne[2]) {
+            auto out = ggml_rope_apply(ctx, x, pe, rope_interleaved);
+            if (ggml_backend_supports_op(backend, out)) {
+                return out;
+            }
+        }
+#endif
         // x: [N, L, n_head, d_head]
         // pe: [L, d_head/2, 2, 2], [[cos, -sin], [sin, cos]]
         int64_t d_head = x->ne[0];
@@ -1126,8 +1138,8 @@ namespace Rope {
         // return: [N, L, n_head*d_head]
         int64_t n_head = q->ne[1];

-        q = apply_rope(ctx->ggml_ctx, q, pe, rope_interleaved);  // [N*n_head, L, d_head]
-        k = apply_rope(ctx->ggml_ctx, k, pe, rope_interleaved);  // [N*n_head, L, d_head]
+        q = apply_rope(ctx, q, pe, rope_interleaved);  // [N*n_head, L, d_head]
+        k = apply_rope(ctx, k, pe, rope_interleaved);  // [N*n_head, L, d_head]

         auto x = ggml_ext_attention_ext(ctx, q, k, v, n_head, mask, true, ctx->flash_attn_enabled, kv_scale);  // [N, L, n_head*d_head]
         return x;
diff --git a/src/model/diffusion/anima.hpp b/src/model/diffusion/anima.hpp
index 9ace46d..a414d96 100644
--- a/src/model/diffusion/anima.hpp
+++ b/src/model/diffusion/anima.hpp
@@ -235,8 +235,8 @@ namespace Anima {
                 if (pe_k == nullptr) {
                     pe_k = pe_q;
                 }
-                auto q_rope = Rope::apply_rope(ctx->ggml_ctx, q4, pe_q, false);
-                auto k_rope = Rope::apply_rope(ctx->ggml_ctx, k4, pe_k, false);
+                auto q_rope = Rope::apply_rope(ctx, q4, pe_q, false);
+                auto k_rope = Rope::apply_rope(ctx, k4, pe_k, false);
                 attn_out    = ggml_ext_attention_ext(ctx,
                                                      q_rope,
                                                      k_rope,
diff --git a/src/model/diffusion/ltxv.hpp b/src/model/diffusion/ltxv.hpp
index 31a65af..e10d5e8 100644
--- a/src/model/diffusion/ltxv.hpp
+++ b/src/model/diffusion/ltxv.hpp
@@ -548,21 +548,22 @@ namespace LTXV {
         return build_rope_matrix_from_frequencies(freqs, dim);
     }

-    __STATIC_INLINE__ ggml_tensor* apply_hidden_rope(ggml_context* ctx,
+    __STATIC_INLINE__ ggml_tensor* apply_hidden_rope(GGMLRunnerContext* runner_ctx,
                                                      ggml_tensor* x,
                                                      ggml_tensor* pe,
                                                      int64_t heads,
                                                      int64_t dim_head,
                                                      bool rope_interleaved) {
+        auto ctx = runner_ctx->ggml_ctx;
         GGML_ASSERT(x->ne[0] == heads * dim_head);
         auto x4 = ggml_reshape_4d(ctx, x, dim_head, heads, x->ne[1], x->ne[2]);
         if (pe != nullptr && pe->ne[3] == x->ne[1] * heads) {
             auto x_flat   = ggml_reshape_4d(ctx, x4, dim_head, 1, x->ne[1] * heads, x->ne[2]);
-            auto out_flat = Rope::apply_rope(ctx, x_flat, pe, rope_interleaved);
+            auto out_flat = Rope::apply_rope(runner_ctx, x_flat, pe, rope_interleaved);
             auto out4     = ggml_reshape_4d(ctx, out_flat, dim_head, heads, x->ne[1], x->ne[2]);
             return ggml_reshape_3d(ctx, out4, heads * dim_head, x->ne[1], x->ne[2]);
         }
-        return Rope::apply_rope(ctx, x4, pe, rope_interleaved);
+        return Rope::apply_rope(runner_ctx, x4, pe, rope_interleaved);
     }

     struct TimestepEmbedder : public GGMLBlock {
@@ -705,8 +706,8 @@ namespace LTXV {
                 if (k_pe == nullptr) {
                     k_pe = pe;
                 }
-                q = apply_hidden_rope(ctx->ggml_ctx, q, pe, heads, dim_head, rope_interleaved);
-                k = apply_hidden_rope(ctx->ggml_ctx, k, k_pe, heads, dim_head, rope_interleaved);
+                q = apply_hidden_rope(ctx, q, pe, heads, dim_head, rope_interleaved);
+                k = apply_hidden_rope(ctx, k, k_pe, heads, dim_head, rope_interleaved);
             }

             auto out = ggml_ext_attention_ext(ctx,
diff --git a/src/model/diffusion/minimax_h3.hpp b/src/model/diffusion/minimax_h3.hpp
index ed7c5ac..7a8693e 100644
--- a/src/model/diffusion/minimax_h3.hpp
+++ b/src/model/diffusion/minimax_h3.hpp
@@ -160,12 +160,13 @@ namespace MiniMaxH3 {
         return ggml_reshape_3d(ctx, x, x->ne[0], x->ne[1], x->ne[2] * x->ne[3]);
     }

-    static ggml_tensor* apply_partial_rope(ggml_context* ctx,
+    static ggml_tensor* apply_partial_rope(GGMLRunnerContext* runner_ctx,
                                            ggml_tensor* x,
                                            ggml_tensor* pe) {
+        auto ctx        = runner_ctx->ggml_ctx;
         int64_t rot_dim = pe->ne[2] * 2;
         GGML_ASSERT(rot_dim <= x->ne[0]);
-        auto rotated = Rope::apply_rope(ctx,
+        auto rotated = Rope::apply_rope(runner_ctx,
                                         ggml_ext_slice(ctx, x, 0, 0, rot_dim),
                                         pe,
                                         false);
@@ -209,8 +210,8 @@ namespace MiniMaxH3 {
             q                = q_norm->forward(ctx, q);
             k                = k_norm->forward(ctx, k);
             if (pe != nullptr) {
-                q = apply_partial_rope(ctx->ggml_ctx, q, pe);
-                k = apply_partial_rope(ctx->ggml_ctx, k, pe);
+                q = apply_partial_rope(ctx, q, pe);
+                k = apply_partial_rope(ctx, k, pe);
             } else {
                 q = attention_layout(ctx->ggml_ctx, q);
                 k = attention_layout(ctx->ggml_ctx, k);
diff --git a/src/model/diffusion/qwen_image_2_1.hpp b/src/model/diffusion/qwen_image_2_1.hpp
index 09311c2..1d37d6a 100644
--- a/src/model/diffusion/qwen_image_2_1.hpp
+++ b/src/model/diffusion/qwen_image_2_1.hpp
@@ -188,8 +188,8 @@ namespace Qwen {
             auto v = project("to_v");
             q      = std::dynamic_pointer_cast<RMSNorm>(blocks["norm_q"])->forward(ctx, q);
             k      = std::dynamic_pointer_cast<RMSNorm>(blocks["norm_k"])->forward(ctx, k);
-            q      = Rope::apply_rope(ctx->ggml_ctx, q, pe);
-            k      = Rope::apply_rope(ctx->ggml_ctx, k, pe);
+            q      = Rope::apply_rope(ctx, q, pe);
+            k      = Rope::apply_rope(ctx, k, pe);
             if (cache.mode == QwenImage21PrefixCache::Mode::STORE) {
                 // Preserve query-first attention evaluation while writing each layer's
                 // prefix before its full-sequence K/V can accumulate across layers.
diff --git a/src/model/vae/minimax_h3_vae.hpp b/src/model/vae/minimax_h3_vae.hpp
index ec6f623..98cf19c 100644
--- a/src/model/vae/minimax_h3_vae.hpp
+++ b/src/model/vae/minimax_h3_vae.hpp
@@ -228,11 +228,12 @@ namespace MiniMaxH3VAE {
         return ggml_reshape_3d(ctx, x, x->ne[0], x->ne[1], x->ne[2] * x->ne[3]);
     }

-    static ggml_tensor* apply_partial_rope(ggml_context* ctx,
+    static ggml_tensor* apply_partial_rope(GGMLRunnerContext* runner_ctx,
                                            ggml_tensor* x,
                                            ggml_tensor* pe) {
+        auto ctx        = runner_ctx->ggml_ctx;
         int64_t rot_dim = pe->ne[2] * 2;
-        auto rotated    = Rope::apply_rope(ctx,
+        auto rotated    = Rope::apply_rope(runner_ctx,
                                            ggml_ext_slice(ctx, x, 0, 0, rot_dim),
                                            pe,
                                            false);
@@ -289,8 +290,8 @@ namespace MiniMaxH3VAE {
                                                   batch_size);
             q                   = ggml_rms_norm(ctx->ggml_ctx, q, 1e-5f);
             k                   = ggml_rms_norm(ctx->ggml_ctx, k, 1e-5f);
-            q                   = apply_partial_rope(ctx->ggml_ctx, q, pe);
-            k                   = apply_partial_rope(ctx->ggml_ctx, k, pe);
+            q                   = apply_partial_rope(ctx, q, pe);
+            k                   = apply_partial_rope(ctx, k, pe);
             auto out            = ggml_ext_attention_ext(ctx,
                                                          q,
                                                          k,