talk-llama : sync llama.cpp

This commit is contained in:
Georgi Gerganov
2026-06-19 12:53:43 +03:00
parent 41cf1278c9
commit 5ed76e9a07
16 changed files with 555 additions and 58 deletions
+1 -1
View File
@@ -1382,7 +1382,7 @@ int llama_context::encode(const llama_batch & batch_inp) {
const auto & hparams = model.hparams;
// eagle3/DFlash: features as encoder input, and non-draft paths fall back to model's input dim
const int64_t n_embd = hparams.n_embd_inp();
const int64_t n_embd = hparams.n_embd_inp_enc();
const int64_t n_vocab = model.vocab.n_tokens();
// note: during encode, we always pass the full sequence starting from pos = 0