diff --git a/src/models/dflash.cpp b/src/models/dflash.cpp index 9b56ac9eca..1e8881c0c7 100644 --- a/src/models/dflash.cpp +++ b/src/models/dflash.cpp @@ -446,19 +446,16 @@ static ggml_tensor * build_dflash2_conv( ggml_tensor * weight_all = ggml_add(ctx0, coeff_all, base_side); + // taps at or past block_size only read the left padding and add nothing + const int64_t n_taps = std::min(kernel_size, block_size); + ggml_tensor * result = nullptr; - for (int64_t tap = 0; tap < kernel_size; ++tap) { + for (int64_t tap = 0; tap < n_taps; ++tap) { ggml_tensor * values = blocks; if (tap > 0) { - ggml_tensor * zeros = ggml_fill(ctx0, - ggml_new_tensor_3d(ctx0, hidden->type, hidden_size, std::min(tap, block_size), n_blocks), 0.0f); - if (tap < block_size) { - ggml_tensor * previous = ggml_view_3d(ctx0, blocks, hidden_size, block_size - tap, n_blocks, - blocks->nb[1], blocks->nb[2], 0); - values = ggml_concat(ctx0, zeros, previous, 1); - } else { - values = zeros; - } + ggml_tensor * previous = ggml_view_3d(ctx0, blocks, hidden_size, block_size - tap, n_blocks, + blocks->nb[1], blocks->nb[2], 0); + values = ggml_pad_ext(ctx0, previous, 0, 0, tap, 0, 0, 0, 0, 0); } values = ggml_reshape_2d(ctx0, values, hidden_size, n_tokens); diff --git a/tools/mtmd/models/conformer.cpp b/tools/mtmd/models/conformer.cpp index 5f2c7b9731..18c3d27bcd 100644 --- a/tools/mtmd/models/conformer.cpp +++ b/tools/mtmd/models/conformer.cpp @@ -124,8 +124,7 @@ ggml_cgraph * clip_graph_conformer::build() { const auto pos_len = matrix_bd->ne[0]; const auto q_len = matrix_bd->ne[1]; const auto h = matrix_bd->ne[2]; - matrix_bd = ggml_pad(ctx0, matrix_bd, 1, 0, 0, 0); - matrix_bd = ggml_roll(ctx0, matrix_bd, 1, 0, 0, 0); + matrix_bd = ggml_pad_ext(ctx0, matrix_bd, 1, 0, 0, 0, 0, 0, 0, 0); matrix_bd = ggml_reshape_3d(ctx0, matrix_bd, q_len, pos_len + 1, h); matrix_bd = ggml_view_3d(ctx0, matrix_bd, q_len, pos_len, h, matrix_bd->nb[1], matrix_bd->nb[2], matrix_bd->nb[0] * q_len); diff --git a/tools/mtmd/models/gemma4a.cpp b/tools/mtmd/models/gemma4a.cpp index 5dd64b7833..f98a8b6fc1 100644 --- a/tools/mtmd/models/gemma4a.cpp +++ b/tools/mtmd/models/gemma4a.cpp @@ -117,14 +117,12 @@ ggml_cgraph * clip_graph_gemma4a::build() { Qcur = ggml_cont(ctx0, ggml_permute(ctx0, Qcur, 0, 3, 1, 2)); // [D, C, B, H] // K/V block context extraction via overlapping view: - // Pad to S*B elements, roll right by P to create left-padding, + // Left pad by P and right pad to S*B elements, // then view with stride C in the block dimension (overlapping windows). auto extract_blocks = [&](ggml_tensor * t) -> ggml_tensor * { - // [D, H, N] -> pad to S*B -> roll right by P -> cont (materialize) + // [D, H, N] -> left pad by P, right pad to S*B const int64_t pad_kv = S * B - n_pos; - t = ggml_pad(ctx0, t, 0, 0, pad_kv, 0); // [D, H, S*B] - t = ggml_roll(ctx0, t, 0, 0, P, 0); // left-pad by P - t = ggml_cont(ctx0, t); // materialize roll (removes view offset) + t = ggml_pad_ext(ctx0, t, 0, 0, 0, 0, P, pad_kv - P, 0, 0); // [D, H, S*B] // Overlapping view: stride for B dim is C positions, not S // ne = [D, H, S, B], data_size = D*H*S*B*sizeof = source_nbytes (exact fit) // nb1=D*sizeof, nb2=D*H*sizeof, nb3=C*D*H*sizeof (overlap: C < S) @@ -219,9 +217,8 @@ ggml_cgraph * clip_graph_gemma4a::build() { x = ggml_cont(ctx0, ggml_transpose(ctx0, x)); } - // Causal depthwise Conv1D via ggml_ssm_conv (pad+roll for left-only padding). - x = ggml_pad(ctx0, x, 4, 0, 0, 0); - x = ggml_roll(ctx0, x, 4, 0, 0, 0); + // Causal depthwise Conv1D via ggml_ssm_conv, left padded only. + x = ggml_pad_ext(ctx0, x, 4, 0, 0, 0, 0, 0, 0, 0); x = ggml_ssm_conv(ctx0, x, layer.conv_dw_w); if (layer.conv_dw_b) { x = ggml_add(ctx0, x, layer.conv_dw_b); diff --git a/tools/mtmd/models/granite-speech.cpp b/tools/mtmd/models/granite-speech.cpp index a158a59ce9..9725def827 100644 --- a/tools/mtmd/models/granite-speech.cpp +++ b/tools/mtmd/models/granite-speech.cpp @@ -143,9 +143,7 @@ ggml_cgraph * clip_graph_granite_speech::build() { } cb(x, "conv_glu", il); - x = ggml_pad(ctx0, x, conv_pad, 0, 0, 0); - x = ggml_roll(ctx0, x, conv_pad, 0, 0, 0); - x = ggml_pad(ctx0, x, conv_pad, 0, 0, 0); + x = ggml_pad_ext(ctx0, x, conv_pad, conv_pad, 0, 0, 0, 0, 0, 0); x = ggml_ssm_conv(ctx0, x, layer.conv_dw_w); cb(x, "conv_dw", il); diff --git a/tools/mtmd/models/parakeet.cpp b/tools/mtmd/models/parakeet.cpp index 8be141d93b..a6d7f07399 100644 --- a/tools/mtmd/models/parakeet.cpp +++ b/tools/mtmd/models/parakeet.cpp @@ -287,8 +287,7 @@ ggml_cgraph * clip_graph_parakeet::build() { const auto n_frame = rel_pos_scores->ne[1]; const auto n_head = rel_pos_scores->ne[2]; - rel_pos_scores = ggml_pad(ctx0, rel_pos_scores, 1, 0, 0, 0); - rel_pos_scores = ggml_roll(ctx0, rel_pos_scores, 1, 0, 0, 0); + rel_pos_scores = ggml_pad_ext(ctx0, rel_pos_scores, 1, 0, 0, 0, 0, 0, 0, 0); rel_pos_scores = ggml_reshape_3d(ctx0, rel_pos_scores, n_frame, pos_window + 1, n_head); rel_pos_scores = ggml_cont(ctx0, rel_pos_scores); @@ -366,9 +365,7 @@ ggml_cgraph * clip_graph_parakeet::build() { // use ggml_ssm_conv for f32 precision const int dw_pad = (hparams.audio_conv_kernel_size - 1) / 2; - cur = ggml_pad(ctx0, cur, dw_pad, 0, 0, 0); - cur = ggml_roll(ctx0, cur, dw_pad, 0, 0, 0); - cur = ggml_pad(ctx0, cur, dw_pad, 0, 0, 0); + cur = ggml_pad_ext(ctx0, cur, dw_pad, dw_pad, 0, 0, 0, 0, 0, 0); ggml_format_name(cur, "enc_%d_conv_dw_pad", il); cur = ggml_ssm_conv(ctx0, cur, layer.conv_dw_w);