llama: add llama_batch_ext (#24669)

* (wip) add llama_batch_ext

* wip

* updated design

* updated impl

* change signature

* unused var

* demo common_prompt_batch_decode

* fix pos

* tmp disable test-batch-alloc

* fix compat

* nits: add const

* no more pos_max

* add comment about llama_batch_ext_set_embd_state

* handle n_embd_out properly

* rename api --> embd_token

* llama_embd

* stub llama_batch_ext_set_embd_state

* support both token + embd + state in batch

* llama_batch_ext_add_embd

* upstream some changes

* nits

* fix test-batch-alloc

* add test for compat
This commit is contained in:
Xuan-Son Nguyen
2026-09-24 16:25:07 +02:00
committed by GitHub
parent 308883b335
commit fc343a84bb
10 changed files with 1210 additions and 262 deletions
+2 -1
View File
@@ -283,7 +283,8 @@ bool llama_hparams::is_ple(uint32_t il) const {
}
uint32_t llama_hparams::n_pos_per_embd() const {
return rope_type == LLAMA_ROPE_TYPE_MROPE || rope_type == LLAMA_ROPE_TYPE_IMROPE ? 4 : 1;
return (rope_type == LLAMA_ROPE_TYPE_MROPE || rope_type == LLAMA_ROPE_TYPE_IMROPE)
? GGML_MROPE_SECTIONS : 1;
}
bool llama_hparams::is_swa(uint32_t il) const {