Commit 57b557cb9 for llama.cpp

commit 57b557cb95a219cb54b5f871d40299e1e33ce707
Author: Pascal <admin@serveurperso.com>
Date:   Mon Sep 28 20:56:15 2026 +0200

    models: pad on the left with ggml_pad_ext (#29567)

    * models: pad on the left with ggml_pad_ext

    The Parakeet, LFM2-Audio, Granite Speech and Gemma 4 audio encoders
    build a left padding as a right pad followed by a roll, and DFlash2
    concatenates a zero filled block in front of the previous tokens.
    ggml_pad_ext does both in one node now that every backend supports a
    left padding. The Gemma 4 audio embeddings are bit identical.

    * models: skip the DFlash2 taps that only read padding

    A tap at or past block_size shifts every row out of the block, so its
    term is zero. The loop runs min(kernel_size, block_size) taps.

diff --git a/src/models/dflash.cpp b/src/models/dflash.cpp
index 9b56ac9ec..1e8881c0c 100644
--- a/src/models/dflash.cpp
+++ b/src/models/dflash.cpp
@@ -446,19 +446,16 @@ static ggml_tensor * build_dflash2_conv(

     ggml_tensor * weight_all = ggml_add(ctx0, coeff_all, base_side);

+    // taps at or past block_size only read the left padding and add nothing
+    const int64_t n_taps = std::min(kernel_size, block_size);
+
     ggml_tensor * result = nullptr;
-    for (int64_t tap = 0; tap < kernel_size; ++tap) {
+    for (int64_t tap = 0; tap < n_taps; ++tap) {
         ggml_tensor * values = blocks;
         if (tap > 0) {
-            ggml_tensor * zeros = ggml_fill(ctx0,
-                    ggml_new_tensor_3d(ctx0, hidden->type, hidden_size, std::min(tap, block_size), n_blocks), 0.0f);
-            if (tap < block_size) {
-                ggml_tensor * previous = ggml_view_3d(ctx0, blocks, hidden_size, block_size - tap, n_blocks,
-                        blocks->nb[1], blocks->nb[2], 0);
-                values = ggml_concat(ctx0, zeros, previous, 1);
-            } else {
-                values = zeros;
-            }
+            ggml_tensor * previous = ggml_view_3d(ctx0, blocks, hidden_size, block_size - tap, n_blocks,
+                    blocks->nb[1], blocks->nb[2], 0);
+            values = ggml_pad_ext(ctx0, previous, 0, 0, tap, 0, 0, 0, 0, 0);
         }
         values = ggml_reshape_2d(ctx0, values, hidden_size, n_tokens);

diff --git a/tools/mtmd/models/conformer.cpp b/tools/mtmd/models/conformer.cpp
index 5f2c7b973..18c3d27bc 100644
--- a/tools/mtmd/models/conformer.cpp
+++ b/tools/mtmd/models/conformer.cpp
@@ -124,8 +124,7 @@ ggml_cgraph * clip_graph_conformer::build() {
                 const auto pos_len = matrix_bd->ne[0];
                 const auto q_len   = matrix_bd->ne[1];
                 const auto h       = matrix_bd->ne[2];
-                matrix_bd          = ggml_pad(ctx0, matrix_bd, 1, 0, 0, 0);
-                matrix_bd          = ggml_roll(ctx0, matrix_bd, 1, 0, 0, 0);
+                matrix_bd          = ggml_pad_ext(ctx0, matrix_bd, 1, 0, 0, 0, 0, 0, 0, 0);
                 matrix_bd          = ggml_reshape_3d(ctx0, matrix_bd, q_len, pos_len + 1, h);
                 matrix_bd          = ggml_view_3d(ctx0, matrix_bd, q_len, pos_len, h, matrix_bd->nb[1],
                                                         matrix_bd->nb[2], matrix_bd->nb[0] * q_len);
diff --git a/tools/mtmd/models/gemma4a.cpp b/tools/mtmd/models/gemma4a.cpp
index 5dd64b783..f98a8b6fc 100644
--- a/tools/mtmd/models/gemma4a.cpp
+++ b/tools/mtmd/models/gemma4a.cpp
@@ -117,14 +117,12 @@ ggml_cgraph * clip_graph_gemma4a::build() {
             Qcur = ggml_cont(ctx0, ggml_permute(ctx0, Qcur, 0, 3, 1, 2)); // [D, C, B, H]

             // K/V block context extraction via overlapping view:
-            // Pad to S*B elements, roll right by P to create left-padding,
+            // Left pad by P and right pad to S*B elements,
             // then view with stride C in the block dimension (overlapping windows).
             auto extract_blocks = [&](ggml_tensor * t) -> ggml_tensor * {
-                // [D, H, N] -> pad to S*B -> roll right by P -> cont (materialize)
+                // [D, H, N] -> left pad by P, right pad to S*B
                 const int64_t pad_kv = S * B - n_pos;
-                t = ggml_pad(ctx0, t, 0, 0, pad_kv, 0);     // [D, H, S*B]
-                t = ggml_roll(ctx0, t, 0, 0, P, 0);          // left-pad by P
-                t = ggml_cont(ctx0, t);                       // materialize roll (removes view offset)
+                t = ggml_pad_ext(ctx0, t, 0, 0, 0, 0, P, pad_kv - P, 0, 0); // [D, H, S*B]
                 // Overlapping view: stride for B dim is C positions, not S
                 // ne = [D, H, S, B], data_size = D*H*S*B*sizeof = source_nbytes (exact fit)
                 // nb1=D*sizeof, nb2=D*H*sizeof, nb3=C*D*H*sizeof (overlap: C < S)
@@ -219,9 +217,8 @@ ggml_cgraph * clip_graph_gemma4a::build() {
                 x = ggml_cont(ctx0, ggml_transpose(ctx0, x));
             }

-            // Causal depthwise Conv1D via ggml_ssm_conv (pad+roll for left-only padding).
-            x = ggml_pad(ctx0, x, 4, 0, 0, 0);
-            x = ggml_roll(ctx0, x, 4, 0, 0, 0);
+            // Causal depthwise Conv1D via ggml_ssm_conv, left padded only.
+            x = ggml_pad_ext(ctx0, x, 4, 0, 0, 0, 0, 0, 0, 0);
             x = ggml_ssm_conv(ctx0, x, layer.conv_dw_w);
             if (layer.conv_dw_b) {
                 x = ggml_add(ctx0, x, layer.conv_dw_b);
diff --git a/tools/mtmd/models/granite-speech.cpp b/tools/mtmd/models/granite-speech.cpp
index a158a59ce..9725def82 100644
--- a/tools/mtmd/models/granite-speech.cpp
+++ b/tools/mtmd/models/granite-speech.cpp
@@ -143,9 +143,7 @@ ggml_cgraph * clip_graph_granite_speech::build() {
             }
             cb(x, "conv_glu", il);

-            x = ggml_pad(ctx0, x, conv_pad, 0, 0, 0);
-            x = ggml_roll(ctx0, x, conv_pad, 0, 0, 0);
-            x = ggml_pad(ctx0, x, conv_pad, 0, 0, 0);
+            x = ggml_pad_ext(ctx0, x, conv_pad, conv_pad, 0, 0, 0, 0, 0, 0);
             x = ggml_ssm_conv(ctx0, x, layer.conv_dw_w);
             cb(x, "conv_dw", il);

diff --git a/tools/mtmd/models/parakeet.cpp b/tools/mtmd/models/parakeet.cpp
index 8be141d93..a6d7f0739 100644
--- a/tools/mtmd/models/parakeet.cpp
+++ b/tools/mtmd/models/parakeet.cpp
@@ -287,8 +287,7 @@ ggml_cgraph * clip_graph_parakeet::build() {
                     const auto n_frame    = rel_pos_scores->ne[1];
                     const auto n_head     = rel_pos_scores->ne[2];

-                    rel_pos_scores = ggml_pad(ctx0, rel_pos_scores, 1, 0, 0, 0);
-                    rel_pos_scores = ggml_roll(ctx0, rel_pos_scores, 1, 0, 0, 0);
+                    rel_pos_scores = ggml_pad_ext(ctx0, rel_pos_scores, 1, 0, 0, 0, 0, 0, 0, 0);

                     rel_pos_scores = ggml_reshape_3d(ctx0, rel_pos_scores, n_frame, pos_window + 1, n_head);
                     rel_pos_scores = ggml_cont(ctx0, rel_pos_scores);
@@ -366,9 +365,7 @@ ggml_cgraph * clip_graph_parakeet::build() {

             // use ggml_ssm_conv for f32 precision
             const int dw_pad = (hparams.audio_conv_kernel_size - 1) / 2;
-            cur = ggml_pad(ctx0, cur, dw_pad, 0, 0, 0);
-            cur = ggml_roll(ctx0, cur, dw_pad, 0, 0, 0);
-            cur = ggml_pad(ctx0, cur, dw_pad, 0, 0, 0);
+            cur = ggml_pad_ext(ctx0, cur, dw_pad, dw_pad, 0, 0, 0, 0, 0, 0);
             ggml_format_name(cur, "enc_%d_conv_dw_pad", il);

             cur = ggml_ssm_conv(ctx0, cur, layer.conv_dw_w);