sync : llama.cpp

This commit is contained in:
Georgi Gerganov
2025-11-17 21:05:46 +02:00
parent 0e5deca8e2
commit b12abefa9b
18 changed files with 511 additions and 16 deletions
+2 -1
View File
@@ -1592,9 +1592,10 @@ ggml_tensor * llm_graph_context::build_attn(
int il) const {
// these nodes are added to the graph together so that they are not reordered
// by doing so, the number of splits in the graph is reduced
// expand k later to enable rope fusion which directly writes into k-v cache
ggml_build_forward_expand(gf, q_cur);
ggml_build_forward_expand(gf, k_cur);
ggml_build_forward_expand(gf, v_cur);
ggml_build_forward_expand(gf, k_cur);
const auto * mctx_cur = inp->mctx;