Commit 57b557cb9 for llama.cpp
commit 57b557cb95a219cb54b5f871d40299e1e33ce707
Author: Pascal <admin@serveurperso.com>
Date: Mon Sep 28 20:56:15 2026 +0200
models: pad on the left with ggml_pad_ext (#29567)
* models: pad on the left with ggml_pad_ext
The Parakeet, LFM2-Audio, Granite Speech and Gemma 4 audio encoders
build a left padding as a right pad followed by a roll, and DFlash2
concatenates a zero filled block in front of the previous tokens.
ggml_pad_ext does both in one node now that every backend supports a
left padding. The Gemma 4 audio embeddings are bit identical.
* models: skip the DFlash2 taps that only read padding
A tap at or past block_size shifts every row out of the block, so its
term is zero. The loop runs min(kernel_size, block_size) taps.
diff --git a/src/models/dflash.cpp b/src/models/dflash.cpp
index 9b56ac9ec..1e8881c0c 100644
--- a/src/models/dflash.cpp
+++ b/src/models/dflash.cpp
@@ -446,19 +446,16 @@ static ggml_tensor * build_dflash2_conv(
ggml_tensor * weight_all = ggml_add(ctx0, coeff_all, base_side);
+ // taps at or past block_size only read the left padding and add nothing
+ const int64_t n_taps = std::min(kernel_size, block_size);
+
ggml_tensor * result = nullptr;
- for (int64_t tap = 0; tap < kernel_size; ++tap) {
+ for (int64_t tap = 0; tap < n_taps; ++tap) {
ggml_tensor * values = blocks;
if (tap > 0) {
- ggml_tensor * zeros = ggml_fill(ctx0,
- ggml_new_tensor_3d(ctx0, hidden->type, hidden_size, std::min(tap, block_size), n_blocks), 0.0f);
- if (tap < block_size) {
- ggml_tensor * previous = ggml_view_3d(ctx0, blocks, hidden_size, block_size - tap, n_blocks,
- blocks->nb[1], blocks->nb[2], 0);
- values = ggml_concat(ctx0, zeros, previous, 1);
- } else {
- values = zeros;
- }
+ ggml_tensor * previous = ggml_view_3d(ctx0, blocks, hidden_size, block_size - tap, n_blocks,
+ blocks->nb[1], blocks->nb[2], 0);
+ values = ggml_pad_ext(ctx0, previous, 0, 0, tap, 0, 0, 0, 0, 0);
}
values = ggml_reshape_2d(ctx0, values, hidden_size, n_tokens);
diff --git a/tools/mtmd/models/conformer.cpp b/tools/mtmd/models/conformer.cpp
index 5f2c7b973..18c3d27bc 100644
--- a/tools/mtmd/models/conformer.cpp
+++ b/tools/mtmd/models/conformer.cpp
@@ -124,8 +124,7 @@ ggml_cgraph * clip_graph_conformer::build() {
const auto pos_len = matrix_bd->ne[0];
const auto q_len = matrix_bd->ne[1];
const auto h = matrix_bd->ne[2];
- matrix_bd = ggml_pad(ctx0, matrix_bd, 1, 0, 0, 0);
- matrix_bd = ggml_roll(ctx0, matrix_bd, 1, 0, 0, 0);
+ matrix_bd = ggml_pad_ext(ctx0, matrix_bd, 1, 0, 0, 0, 0, 0, 0, 0);
matrix_bd = ggml_reshape_3d(ctx0, matrix_bd, q_len, pos_len + 1, h);
matrix_bd = ggml_view_3d(ctx0, matrix_bd, q_len, pos_len, h, matrix_bd->nb[1],
matrix_bd->nb[2], matrix_bd->nb[0] * q_len);
diff --git a/tools/mtmd/models/gemma4a.cpp b/tools/mtmd/models/gemma4a.cpp
index 5dd64b783..f98a8b6fc 100644
--- a/tools/mtmd/models/gemma4a.cpp
+++ b/tools/mtmd/models/gemma4a.cpp
@@ -117,14 +117,12 @@ ggml_cgraph * clip_graph_gemma4a::build() {
Qcur = ggml_cont(ctx0, ggml_permute(ctx0, Qcur, 0, 3, 1, 2)); // [D, C, B, H]
// K/V block context extraction via overlapping view:
- // Pad to S*B elements, roll right by P to create left-padding,
+ // Left pad by P and right pad to S*B elements,
// then view with stride C in the block dimension (overlapping windows).
auto extract_blocks = [&](ggml_tensor * t) -> ggml_tensor * {
- // [D, H, N] -> pad to S*B -> roll right by P -> cont (materialize)
+ // [D, H, N] -> left pad by P, right pad to S*B
const int64_t pad_kv = S * B - n_pos;
- t = ggml_pad(ctx0, t, 0, 0, pad_kv, 0); // [D, H, S*B]
- t = ggml_roll(ctx0, t, 0, 0, P, 0); // left-pad by P
- t = ggml_cont(ctx0, t); // materialize roll (removes view offset)
+ t = ggml_pad_ext(ctx0, t, 0, 0, 0, 0, P, pad_kv - P, 0, 0); // [D, H, S*B]
// Overlapping view: stride for B dim is C positions, not S
// ne = [D, H, S, B], data_size = D*H*S*B*sizeof = source_nbytes (exact fit)
// nb1=D*sizeof, nb2=D*H*sizeof, nb3=C*D*H*sizeof (overlap: C < S)
@@ -219,9 +217,8 @@ ggml_cgraph * clip_graph_gemma4a::build() {
x = ggml_cont(ctx0, ggml_transpose(ctx0, x));
}
- // Causal depthwise Conv1D via ggml_ssm_conv (pad+roll for left-only padding).
- x = ggml_pad(ctx0, x, 4, 0, 0, 0);
- x = ggml_roll(ctx0, x, 4, 0, 0, 0);
+ // Causal depthwise Conv1D via ggml_ssm_conv, left padded only.
+ x = ggml_pad_ext(ctx0, x, 4, 0, 0, 0, 0, 0, 0, 0);
x = ggml_ssm_conv(ctx0, x, layer.conv_dw_w);
if (layer.conv_dw_b) {
x = ggml_add(ctx0, x, layer.conv_dw_b);
diff --git a/tools/mtmd/models/granite-speech.cpp b/tools/mtmd/models/granite-speech.cpp
index a158a59ce..9725def82 100644
--- a/tools/mtmd/models/granite-speech.cpp
+++ b/tools/mtmd/models/granite-speech.cpp
@@ -143,9 +143,7 @@ ggml_cgraph * clip_graph_granite_speech::build() {
}
cb(x, "conv_glu", il);
- x = ggml_pad(ctx0, x, conv_pad, 0, 0, 0);
- x = ggml_roll(ctx0, x, conv_pad, 0, 0, 0);
- x = ggml_pad(ctx0, x, conv_pad, 0, 0, 0);
+ x = ggml_pad_ext(ctx0, x, conv_pad, conv_pad, 0, 0, 0, 0, 0, 0);
x = ggml_ssm_conv(ctx0, x, layer.conv_dw_w);
cb(x, "conv_dw", il);
diff --git a/tools/mtmd/models/parakeet.cpp b/tools/mtmd/models/parakeet.cpp
index 8be141d93..a6d7f0739 100644
--- a/tools/mtmd/models/parakeet.cpp
+++ b/tools/mtmd/models/parakeet.cpp
@@ -287,8 +287,7 @@ ggml_cgraph * clip_graph_parakeet::build() {
const auto n_frame = rel_pos_scores->ne[1];
const auto n_head = rel_pos_scores->ne[2];
- rel_pos_scores = ggml_pad(ctx0, rel_pos_scores, 1, 0, 0, 0);
- rel_pos_scores = ggml_roll(ctx0, rel_pos_scores, 1, 0, 0, 0);
+ rel_pos_scores = ggml_pad_ext(ctx0, rel_pos_scores, 1, 0, 0, 0, 0, 0, 0, 0);
rel_pos_scores = ggml_reshape_3d(ctx0, rel_pos_scores, n_frame, pos_window + 1, n_head);
rel_pos_scores = ggml_cont(ctx0, rel_pos_scores);
@@ -366,9 +365,7 @@ ggml_cgraph * clip_graph_parakeet::build() {
// use ggml_ssm_conv for f32 precision
const int dw_pad = (hparams.audio_conv_kernel_size - 1) / 2;
- cur = ggml_pad(ctx0, cur, dw_pad, 0, 0, 0);
- cur = ggml_roll(ctx0, cur, dw_pad, 0, 0, 0);
- cur = ggml_pad(ctx0, cur, dw_pad, 0, 0, 0);
+ cur = ggml_pad_ext(ctx0, cur, dw_pad, dw_pad, 0, 0, 0, 0, 0, 0);
ggml_format_name(cur, "enc_%d_conv_dw_pad", il);
cur = ggml_ssm_conv(ctx0, cur, layer.conv_dw_w);