mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-03 03:17:32 -05:00
models: pad on the left with ggml_pad_ext (#29567)
* models: pad on the left with ggml_pad_ext The Parakeet, LFM2-Audio, Granite Speech and Gemma 4 audio encoders build a left padding as a right pad followed by a roll, and DFlash2 concatenates a zero filled block in front of the previous tokens. ggml_pad_ext does both in one node now that every backend supports a left padding. The Gemma 4 audio embeddings are bit identical. * models: skip the DFlash2 taps that only read padding A tap at or past block_size shifts every row out of the block, so its term is zero. The loop runs min(kernel_size, block_size) taps.
This commit is contained in:
+7
-10
@@ -446,19 +446,16 @@ static ggml_tensor * build_dflash2_conv(
|
||||
|
||||
ggml_tensor * weight_all = ggml_add(ctx0, coeff_all, base_side);
|
||||
|
||||
// taps at or past block_size only read the left padding and add nothing
|
||||
const int64_t n_taps = std::min(kernel_size, block_size);
|
||||
|
||||
ggml_tensor * result = nullptr;
|
||||
for (int64_t tap = 0; tap < kernel_size; ++tap) {
|
||||
for (int64_t tap = 0; tap < n_taps; ++tap) {
|
||||
ggml_tensor * values = blocks;
|
||||
if (tap > 0) {
|
||||
ggml_tensor * zeros = ggml_fill(ctx0,
|
||||
ggml_new_tensor_3d(ctx0, hidden->type, hidden_size, std::min(tap, block_size), n_blocks), 0.0f);
|
||||
if (tap < block_size) {
|
||||
ggml_tensor * previous = ggml_view_3d(ctx0, blocks, hidden_size, block_size - tap, n_blocks,
|
||||
blocks->nb[1], blocks->nb[2], 0);
|
||||
values = ggml_concat(ctx0, zeros, previous, 1);
|
||||
} else {
|
||||
values = zeros;
|
||||
}
|
||||
ggml_tensor * previous = ggml_view_3d(ctx0, blocks, hidden_size, block_size - tap, n_blocks,
|
||||
blocks->nb[1], blocks->nb[2], 0);
|
||||
values = ggml_pad_ext(ctx0, previous, 0, 0, tap, 0, 0, 0, 0, 0);
|
||||
}
|
||||
values = ggml_reshape_2d(ctx0, values, hidden_size, n_tokens);
|
||||
|
||||
|
||||
@@ -124,8 +124,7 @@ ggml_cgraph * clip_graph_conformer::build() {
|
||||
const auto pos_len = matrix_bd->ne[0];
|
||||
const auto q_len = matrix_bd->ne[1];
|
||||
const auto h = matrix_bd->ne[2];
|
||||
matrix_bd = ggml_pad(ctx0, matrix_bd, 1, 0, 0, 0);
|
||||
matrix_bd = ggml_roll(ctx0, matrix_bd, 1, 0, 0, 0);
|
||||
matrix_bd = ggml_pad_ext(ctx0, matrix_bd, 1, 0, 0, 0, 0, 0, 0, 0);
|
||||
matrix_bd = ggml_reshape_3d(ctx0, matrix_bd, q_len, pos_len + 1, h);
|
||||
matrix_bd = ggml_view_3d(ctx0, matrix_bd, q_len, pos_len, h, matrix_bd->nb[1],
|
||||
matrix_bd->nb[2], matrix_bd->nb[0] * q_len);
|
||||
|
||||
@@ -117,14 +117,12 @@ ggml_cgraph * clip_graph_gemma4a::build() {
|
||||
Qcur = ggml_cont(ctx0, ggml_permute(ctx0, Qcur, 0, 3, 1, 2)); // [D, C, B, H]
|
||||
|
||||
// K/V block context extraction via overlapping view:
|
||||
// Pad to S*B elements, roll right by P to create left-padding,
|
||||
// Left pad by P and right pad to S*B elements,
|
||||
// then view with stride C in the block dimension (overlapping windows).
|
||||
auto extract_blocks = [&](ggml_tensor * t) -> ggml_tensor * {
|
||||
// [D, H, N] -> pad to S*B -> roll right by P -> cont (materialize)
|
||||
// [D, H, N] -> left pad by P, right pad to S*B
|
||||
const int64_t pad_kv = S * B - n_pos;
|
||||
t = ggml_pad(ctx0, t, 0, 0, pad_kv, 0); // [D, H, S*B]
|
||||
t = ggml_roll(ctx0, t, 0, 0, P, 0); // left-pad by P
|
||||
t = ggml_cont(ctx0, t); // materialize roll (removes view offset)
|
||||
t = ggml_pad_ext(ctx0, t, 0, 0, 0, 0, P, pad_kv - P, 0, 0); // [D, H, S*B]
|
||||
// Overlapping view: stride for B dim is C positions, not S
|
||||
// ne = [D, H, S, B], data_size = D*H*S*B*sizeof = source_nbytes (exact fit)
|
||||
// nb1=D*sizeof, nb2=D*H*sizeof, nb3=C*D*H*sizeof (overlap: C < S)
|
||||
@@ -219,9 +217,8 @@ ggml_cgraph * clip_graph_gemma4a::build() {
|
||||
x = ggml_cont(ctx0, ggml_transpose(ctx0, x));
|
||||
}
|
||||
|
||||
// Causal depthwise Conv1D via ggml_ssm_conv (pad+roll for left-only padding).
|
||||
x = ggml_pad(ctx0, x, 4, 0, 0, 0);
|
||||
x = ggml_roll(ctx0, x, 4, 0, 0, 0);
|
||||
// Causal depthwise Conv1D via ggml_ssm_conv, left padded only.
|
||||
x = ggml_pad_ext(ctx0, x, 4, 0, 0, 0, 0, 0, 0, 0);
|
||||
x = ggml_ssm_conv(ctx0, x, layer.conv_dw_w);
|
||||
if (layer.conv_dw_b) {
|
||||
x = ggml_add(ctx0, x, layer.conv_dw_b);
|
||||
|
||||
@@ -143,9 +143,7 @@ ggml_cgraph * clip_graph_granite_speech::build() {
|
||||
}
|
||||
cb(x, "conv_glu", il);
|
||||
|
||||
x = ggml_pad(ctx0, x, conv_pad, 0, 0, 0);
|
||||
x = ggml_roll(ctx0, x, conv_pad, 0, 0, 0);
|
||||
x = ggml_pad(ctx0, x, conv_pad, 0, 0, 0);
|
||||
x = ggml_pad_ext(ctx0, x, conv_pad, conv_pad, 0, 0, 0, 0, 0, 0);
|
||||
x = ggml_ssm_conv(ctx0, x, layer.conv_dw_w);
|
||||
cb(x, "conv_dw", il);
|
||||
|
||||
|
||||
@@ -287,8 +287,7 @@ ggml_cgraph * clip_graph_parakeet::build() {
|
||||
const auto n_frame = rel_pos_scores->ne[1];
|
||||
const auto n_head = rel_pos_scores->ne[2];
|
||||
|
||||
rel_pos_scores = ggml_pad(ctx0, rel_pos_scores, 1, 0, 0, 0);
|
||||
rel_pos_scores = ggml_roll(ctx0, rel_pos_scores, 1, 0, 0, 0);
|
||||
rel_pos_scores = ggml_pad_ext(ctx0, rel_pos_scores, 1, 0, 0, 0, 0, 0, 0, 0);
|
||||
|
||||
rel_pos_scores = ggml_reshape_3d(ctx0, rel_pos_scores, n_frame, pos_window + 1, n_head);
|
||||
rel_pos_scores = ggml_cont(ctx0, rel_pos_scores);
|
||||
@@ -366,9 +365,7 @@ ggml_cgraph * clip_graph_parakeet::build() {
|
||||
|
||||
// use ggml_ssm_conv for f32 precision
|
||||
const int dw_pad = (hparams.audio_conv_kernel_size - 1) / 2;
|
||||
cur = ggml_pad(ctx0, cur, dw_pad, 0, 0, 0);
|
||||
cur = ggml_roll(ctx0, cur, dw_pad, 0, 0, 0);
|
||||
cur = ggml_pad(ctx0, cur, dw_pad, 0, 0, 0);
|
||||
cur = ggml_pad_ext(ctx0, cur, dw_pad, dw_pad, 0, 0, 0, 0, 0, 0);
|
||||
ggml_format_name(cur, "enc_%d_conv_dw_pad", il);
|
||||
|
||||
cur = ggml_ssm_conv(ctx0, cur, layer.conv_dw_w);
|
||||
|
||||
Reference in New Issue
Block a user