Commit b9acf138a for llama.cpp
commit b9acf138a1e28ce1fc23b5a4fc4b12444b50f7ea
Author: Piotr Wilkin (ilintar) <piotr.wilkin@syndatis.com>
Date: Wed Oct 7 13:16:53 2026 +0200
feat: add GLM5Next MTP, optimize (#29928)
* llama : add GLM5-Next NextN (MTP) graph
Build the GLM5-Next multi-token-prediction head as graph_mtp: the NextN block
embeds enorm(tok)+hnorm(h) through eh_proj, runs one plain DSA layer and the
shared lm_head, reusing the trunk's builders through the no_build tag ctor.
llama_memory_recurrent also tolerates a partial seq_rm when the context holds
no recurrent layers, which is what the MTP draft context needs.
Assisted-by: Claude
* llama : glm5-next: skip dead compute in headless NextN forwards
A NextN forward with no output rows (the MTP catch-up and the draft-context
prefill) persists only through its cache writes, so the headless graph keeps
the MLA latent, indexer key|gate and pooled-key writes and drops the query
path, the indexer selection, the attention body, the FFN and the LM head. The
4-token catch-up falls from 6.9 ms to 0.33 ms of kernels; the greedy output
hashes and the draft acceptance are unchanged.
Assisted-by: Claude
* llama : glm5-next: fix NextN extraction contracts and shared-tail rollback
Three fixes from the architectural review. The headless graph prune now also
requires that no unmasked nextn extraction is live, because that mode reads
n_tokens hidden rows regardless of the logits flags. Masked extraction
publishes the hidden rows gathered by the output ids, so a batch whose output
flags are not a prefix exports the right rows. A partial recurrent rollback
whose tail cell is shared with another sequence is now rejected instead of
silently moving that sequence's tail.
Assisted-by: Claude
* llama : glm5-next: tidy comments in the MTP changes
Assisted-by: Claude
* llama : glm5-next: crop the MTP graph to the output rows instead of pruning it
Replace the headless NextN prune with the crop pattern the other MTP
graphs use: gather the attention output and the block input at the
output ids before the position-wise FFN and the shared head. A NextN
forward with no output rows (the MTP catch-up and the draft-context
prefill) then runs the FFN and the head over zero rows. The 4-token
catch-up falls from 6.9 ms to 2.9 ms of kernels; greedy output hashes
are unchanged.
Assisted-by: Claude
* glm5-next: use the nextn crop helpers in the MTP graph
Replace the local crop condition and the masked select of t_h_nextn
with crop_before_nextn and crop_after_nextn, so the MTP graph narrows
its rows the same way as the main graph and the other models.
Describe the shared cell and empty filter branches of the recurrent
partial rollback.
* glm5-next: load MTP-only and trunk-only GGUF files
Make the trunk tensors optional when the file only holds the NextN
layer, and the NextN tensors optional when the file only holds the
trunk, so the split MTP GGUF loads as a draft model.
Co-authored-by: Georgi Gerganov <ggerganov@gmail.com>
---------
Co-authored-by: Pascal <admin@serveurperso.com>
Co-authored-by: Georgi Gerganov <ggerganov@gmail.com>
diff --git a/src/llama-memory-recurrent.cpp b/src/llama-memory-recurrent.cpp
index 53130e868..2893d819e 100644
--- a/src/llama-memory-recurrent.cpp
+++ b/src/llama-memory-recurrent.cpp
@@ -199,6 +199,10 @@ bool llama_memory_recurrent::seq_rm(llama_seq_id seq_id, llama_pos p0, llama_pos
// partial rollback via per-token snapshot index (bounded by n_rs_seq)
if (0 < p0 && p0 <= cell.pos && p1 > cell.pos) {
+ // a cell shared by several sequences cannot move back for only one of them
+ if (cell.seq_id.size() > 1) {
+ return false;
+ }
// the filter kept no layer (e.g. an MTP draft context), so only the position moves back
if (is_empty()) {
cell.pos = p0 - 1;
diff --git a/src/models/glm5-next.cpp b/src/models/glm5-next.cpp
index 955405c75..4343e6ac6 100644
--- a/src/models/glm5-next.cpp
+++ b/src/models/glm5-next.cpp
@@ -64,9 +64,13 @@ void llama_model_glm5_next::load_arch_tensors(llama_model_loader & ml) {
const int64_t hc = hparams.dsv4_hc_mult;
const int64_t hc_mix_dim = (2 + hc)*hc;
- // the NextN block is loaded but only used by the MTP graph.
- // Separated trunk_only/mtp_only handling TODO with DECODER_MTP graph in the MTP follow up
- int mtp_flags = 0;
+ // TODO: consolidate the trunk_only/mtp_only detection and flags shared with glm4-moe and qwen35moe
+ const bool mtp_only = (hparams.n_layer_nextn > 0) && (ml.get_weight("blk.0.attn_norm.weight") == nullptr);
+ const std::string mtp_probe = "blk." + std::to_string(n_layer) + ".nextn.eh_proj.weight";
+ const bool trunk_only = (hparams.n_layer_nextn > 0) && (ml.get_weight(mtp_probe.c_str()) == nullptr);
+ const int trunk_flags = mtp_only ? TENSOR_NOT_REQUIRED : 0;
+ int mtp_flags = trunk_only ? TENSOR_NOT_REQUIRED : 0;
+
if (!ml.load_mtp) {
mtp_flags |= TENSOR_SKIP;
}
@@ -79,18 +83,19 @@ void llama_model_glm5_next::load_arch_tensors(llama_model_loader & ml) {
for (int i = 0; i < n_layer_all; ++i) {
auto & layer = layers[i];
- const int flags = (i >= n_layer) ? mtp_flags : 0;
+ const bool is_nextn = i >= n_layer;
+ const int flags = is_nextn ? mtp_flags : trunk_flags;
layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, flags);
layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), {n_embd}, flags);
if (i < n_layer) {
- layer.hc_attn_fn = create_tensor(tn(LLM_TENSOR_HC_ATTN_FN, "weight", i), {hc*n_embd, hc_mix_dim}, 0);
- layer.hc_attn_base = create_tensor(tn(LLM_TENSOR_HC_ATTN_BASE, "weight", i), {hc_mix_dim}, 0);
- layer.hc_attn_scale = create_tensor(tn(LLM_TENSOR_HC_ATTN_SCALE, "weight", i), {3}, 0);
- layer.hc_ffn_fn = create_tensor(tn(LLM_TENSOR_HC_FFN_FN, "weight", i), {hc*n_embd, hc_mix_dim}, 0);
- layer.hc_ffn_base = create_tensor(tn(LLM_TENSOR_HC_FFN_BASE, "weight", i), {hc_mix_dim}, 0);
- layer.hc_ffn_scale = create_tensor(tn(LLM_TENSOR_HC_FFN_SCALE, "weight", i), {3}, 0);
+ layer.hc_attn_fn = create_tensor(tn(LLM_TENSOR_HC_ATTN_FN, "weight", i), {hc*n_embd, hc_mix_dim}, flags);
+ layer.hc_attn_base = create_tensor(tn(LLM_TENSOR_HC_ATTN_BASE, "weight", i), {hc_mix_dim}, flags);
+ layer.hc_attn_scale = create_tensor(tn(LLM_TENSOR_HC_ATTN_SCALE, "weight", i), {3}, flags);
+ layer.hc_ffn_fn = create_tensor(tn(LLM_TENSOR_HC_FFN_FN, "weight", i), {hc*n_embd, hc_mix_dim}, flags);
+ layer.hc_ffn_base = create_tensor(tn(LLM_TENSOR_HC_FFN_BASE, "weight", i), {hc_mix_dim}, flags);
+ layer.hc_ffn_scale = create_tensor(tn(LLM_TENSOR_HC_FFN_SCALE, "weight", i), {3}, flags);
}
const int64_t head_dim = hparams.n_embd_head_kda;
@@ -100,25 +105,25 @@ void llama_model_glm5_next::load_arch_tensors(llama_model_loader & ml) {
if (hparams.is_recr(i)) {
auto conv = [&](llm_tensor tid) {
ggml_tensor * t = create_tensor(tn(tid, "weight", i), {d_conv, 1, d_inner, 1}, TENSOR_NOT_REQUIRED);
- return t ? t : create_tensor(tn(tid, "weight", i), {d_conv, 1, d_inner}, 0);
+ return t ? t : create_tensor(tn(tid, "weight", i), {d_conv, 1, d_inner}, flags);
};
layer.ssm_q_conv = conv(LLM_TENSOR_SSM_CONV1D_Q);
layer.ssm_k_conv = conv(LLM_TENSOR_SSM_CONV1D_K);
layer.ssm_v_conv = conv(LLM_TENSOR_SSM_CONV1D_V);
- create_tensor_qkv(layer, i, n_embd, d_inner, d_inner, d_inner, 0);
+ create_tensor_qkv(layer, i, n_embd, d_inner, d_inner, d_inner, flags);
- layer.ssm_f_a = create_tensor(tn(LLM_TENSOR_SSM_F_A, "weight", i), {n_embd, head_dim}, 0);
- layer.ssm_f_b = create_tensor(tn(LLM_TENSOR_SSM_F_B, "weight", i), {head_dim, d_inner}, 0);
- layer.ssm_beta = create_tensor(tn(LLM_TENSOR_SSM_BETA, "weight", i), {n_embd, n_head}, 0);
+ layer.ssm_f_a = create_tensor(tn(LLM_TENSOR_SSM_F_A, "weight", i), {n_embd, head_dim}, flags);
+ layer.ssm_f_b = create_tensor(tn(LLM_TENSOR_SSM_F_B, "weight", i), {head_dim, d_inner}, flags);
+ layer.ssm_beta = create_tensor(tn(LLM_TENSOR_SSM_BETA, "weight", i), {n_embd, n_head}, flags);
- layer.ssm_a = create_tensor(tn(LLM_TENSOR_SSM_A_NOSCAN, i), {n_head}, 0);
- layer.ssm_dt_b = create_tensor(tn(LLM_TENSOR_SSM_DT, "bias", i), {d_inner}, 0);
+ layer.ssm_a = create_tensor(tn(LLM_TENSOR_SSM_A_NOSCAN, i), {n_head}, flags);
+ layer.ssm_dt_b = create_tensor(tn(LLM_TENSOR_SSM_DT, "bias", i), {d_inner}, flags);
- layer.ssm_g_a = create_tensor(tn(LLM_TENSOR_SSM_G_A, "weight", i), {n_embd, head_dim}, 0);
- layer.ssm_g_b = create_tensor(tn(LLM_TENSOR_SSM_G_B, "weight", i), {head_dim, d_inner}, 0);
- layer.ssm_o_norm = create_tensor(tn(LLM_TENSOR_SSM_NORM, "weight", i), {head_dim}, 0);
- layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", i), {d_inner, n_embd}, 0);
+ layer.ssm_g_a = create_tensor(tn(LLM_TENSOR_SSM_G_A, "weight", i), {n_embd, head_dim}, flags);
+ layer.ssm_g_b = create_tensor(tn(LLM_TENSOR_SSM_G_B, "weight", i), {head_dim, d_inner}, flags);
+ layer.ssm_o_norm = create_tensor(tn(LLM_TENSOR_SSM_NORM, "weight", i), {head_dim}, flags);
+ layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", i), {d_inner, n_embd}, flags);
} else {
const int64_t q_lora_rank = hparams.n_lora_q;
const int64_t kv_lora_rank = hparams.n_lora_kv;
@@ -155,9 +160,9 @@ void llama_model_glm5_next::load_arch_tensors(llama_model_loader & ml) {
}
if (i < (int) hparams.n_layer_dense_lead) {
- layer.ffn_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "weight", i), {n_embd, n_ff}, 0);
- layer.ffn_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "weight", i), {n_ff, n_embd}, 0);
- layer.ffn_up = create_tensor(tn(LLM_TENSOR_FFN_UP, "weight", i), {n_embd, n_ff}, 0);
+ layer.ffn_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "weight", i), {n_embd, n_ff}, flags);
+ layer.ffn_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "weight", i), {n_ff, n_embd}, flags);
+ layer.ffn_up = create_tensor(tn(LLM_TENSOR_FFN_UP, "weight", i), {n_embd, n_ff}, flags);
} else {
const int64_t n_ff_exp = hparams.n_ff_exp(i);
const int64_t n_expert_shared = hparams.n_expert_shared;
@@ -187,7 +192,7 @@ void llama_model_glm5_next::load_arch_tensors(llama_model_loader & ml) {
std::unique_ptr<llm_graph_context> llama_model_glm5_next::build_arch_graph(const llm_graph_params & params) const {
if (params.gtype == LLM_GRAPH_TYPE_DECODER_MTP) {
- throw std::runtime_error("GLM5-Next NextN graph not implemented yet");
+ return std::make_unique<graph_mtp>(*this, params);
}
return std::make_unique<graph>(*this, params);
}
@@ -537,6 +542,136 @@ ggml_tensor * llama_model_glm5_next::graph::build_hc_post(
return out;
}
+// construct the graph helpers without building the trunk, so graph_mtp can reuse them
+llama_model_glm5_next::graph::graph(const llama_model & model, const llm_graph_params & params, no_build) :
+ llm_build_delta_net_base(params), model(model) {
+}
+
+llama_model_glm5_next::graph_mtp::graph_mtp(const llama_model & model, const llm_graph_params & params) :
+ graph(model, params, no_build{}) {
+
+ GGML_ASSERT(hparams.n_layer_nextn > 0 && "GLM5-Next MTP requires n_layer_nextn > 0");
+ GGML_ASSERT(hparams.n_layer_nextn == 1 && "GLM5-Next MTP currently supports a single NextN block");
+
+ const int il = hparams.n_layer() + cparams.nextn_layer_offset;
+ GGML_ASSERT(cparams.nextn_layer_offset >= 0 &&
+ cparams.nextn_layer_offset < (int) hparams.n_layer_nextn &&
+ "nextn_layer_offset out of range [0, n_layer_nextn)");
+
+ const auto & layer = model.layers[il];
+
+ GGML_ASSERT(layer.nextn.eh_proj && layer.nextn.enorm && layer.nextn.hnorm &&
+ "GLM5-Next MTP block is missing - load the model with MTP enabled");
+
+ GGML_ASSERT(hparams.n_embd_out() == (uint32_t) n_embd && "GLM5-Next MTP hidden width mismatch");
+
+ const auto * mctx_hyb = static_cast<const llama_memory_hybrid_idx_context *>(mctx);
+
+ auto * inp_hyb = build_inp_mem_hybrid_k();
+ auto * inp_attn = inp_hyb->get_attn();
+ auto * inp_kpool = build_inp_kpool(mctx_hyb);
+
+ ggml_build_forward_expand(gf, inp_hyb->get_recr()->s_copy);
+
+ ggml_tensor * inp_out_ids = build_inp_out_ids();
+
+ auto inp = std::make_unique<llm_graph_input_embd_h>(n_embd);
+
+ inp->tokens = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_tokens);
+ ggml_set_input(inp->tokens);
+
+ inp->embd = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_embd, n_tokens);
+ ggml_set_input(inp->embd);
+
+ inp->h = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_embd, n_tokens);
+ ggml_set_input(inp->h);
+ ggml_set_name(inp->h, "mtp_h_input");
+
+ ggml_tensor * tok_embd = ggml_get_rows(ctx0,
+ layer.nextn.embed_tokens ? layer.nextn.embed_tokens : model.tok_embd, inp->tokens);
+ cb(tok_embd, "mtp_tok_embd", il);
+
+ ggml_tensor * h = inp->h;
+
+ res->add_input(std::move(inp));
+
+ ggml_tensor * e_norm = build_norm(tok_embd, layer.nextn.enorm, nullptr, LLM_NORM_RMS, il);
+ cb(e_norm, "mtp_enorm", il);
+
+ ggml_tensor * h_norm = build_norm(h, layer.nextn.hnorm, nullptr, LLM_NORM_RMS, il);
+ cb(h_norm, "mtp_hnorm", il);
+
+ ggml_tensor * cur = build_lora_mm(layer.nextn.eh_proj, ggml_concat(ctx0, e_norm, h_norm, 0), layer.nextn.eh_proj_s);
+ cb(cur, "mtp_eh_proj", il);
+
+ ggml_tensor * inpSA = cur;
+
+ cur = build_norm(cur, layer.attn_norm, nullptr, LLM_NORM_RMS, il);
+ cb(cur, "mtp_attn_norm", il);
+
+ ggml_tensor * prev_sel = nullptr;
+ cur = build_dsa_layer(cur, layer, mctx_hyb, inp_attn, inp_kpool, &prev_sel, il);
+ cb(cur, "mtp_attn_out", il);
+
+ // narrow to the output tokens before the position-wise FFN; unmasked nextn embeddings need all rows
+ if (crop_before_nextn(inp_out_ids)) {
+ cur = ggml_get_rows(ctx0, cur, inp_out_ids);
+ inpSA = ggml_get_rows(ctx0, inpSA, inp_out_ids);
+ }
+
+ ggml_tensor * ffn_inp = ggml_add(ctx0, cur, inpSA);
+ cb(ffn_inp, "mtp_ffn_inp", il);
+
+ cur = build_norm(ffn_inp, layer.ffn_norm, nullptr, LLM_NORM_RMS, il);
+ cb(cur, "mtp_ffn_norm", il);
+
+ ggml_tensor * moe_out = build_moe_ffn(cur,
+ layer.ffn_gate_inp,
+ layer.ffn_up_exps,
+ layer.ffn_gate_exps,
+ layer.ffn_down_exps,
+ layer.ffn_exp_probs_b,
+ n_expert, n_expert_used,
+ LLM_FFN_SILU, hparams.expert_weights_norm,
+ hparams.expert_weights_scale,
+ (llama_expert_gating_func_type) hparams.expert_gating_func,
+ il);
+ cb(moe_out, "mtp_ffn_moe_out", il);
+
+ ggml_tensor * ffn_shexp = build_ffn(cur,
+ layer.ffn_up_shexp, nullptr, nullptr,
+ layer.ffn_gate_shexp, nullptr, nullptr,
+ layer.ffn_down_shexp, nullptr, nullptr,
+ nullptr, LLM_FFN_SILU, LLM_FFN_PAR, il);
+ cb(ffn_shexp, "mtp_ffn_shexp", il);
+
+ cur = ggml_add(ctx0, moe_out, ffn_shexp);
+ cb(cur, "mtp_ffn_out", il);
+
+ cur = ggml_add(ctx0, cur, ffn_inp);
+ cb(cur, "mtp_post_ffn", il);
+
+ ggml_tensor * head_norm = layer.nextn.shared_head_norm ? layer.nextn.shared_head_norm : model.output_norm;
+ GGML_ASSERT(head_norm && "GLM5-Next MTP: missing both nextn.shared_head_norm and output_norm");
+ cur = build_norm(cur, head_norm, nullptr, LLM_NORM_RMS, -1);
+ cb(cur, "h_nextn", -1);
+ res->t_h_nextn = cur;
+
+ if (crop_after_nextn(inp_out_ids)) {
+ cur = ggml_get_rows(ctx0, cur, inp_out_ids);
+ }
+ cb(cur, "mtp_shared_head_norm", -1);
+
+ ggml_tensor * head_w = layer.nextn.shared_head_head ? layer.nextn.shared_head_head : model.output;
+ ggml_tensor * head_s = layer.nextn.shared_head_head ? layer.nextn.shared_head_head_s : model.output_s;
+ GGML_ASSERT(head_w && "GLM5-Next MTP: missing both nextn.shared_head_head and output");
+ cur = build_lora_mm(head_w, cur, head_s);
+ cb(cur, "result_output", -1);
+
+ res->t_logits = cur;
+ ggml_build_forward_expand(gf, cur);
+}
+
llama_model_glm5_next::graph::graph(const llama_model & model, const llm_graph_params & params) :
llm_build_delta_net_base(params), model(model) {
diff --git a/src/models/models.h b/src/models/models.h
index 023ed3021..1ef0c5156 100644
--- a/src/models/models.h
+++ b/src/models/models.h
@@ -2741,6 +2741,10 @@ struct llama_model_glm5_next : public llama_model_base {
struct graph : public llm_build_delta_net_base {
graph(const llama_model & model, const llm_graph_params & params);
+ // build the helpers without the trunk, so graph_mtp can reuse them
+ struct no_build {};
+ graph(const llama_model & model, const llm_graph_params & params, no_build);
+
// collapse the hc streams with per-stream weights
ggml_tensor * build_hc_pre(
ggml_tensor * x,
@@ -2789,6 +2793,11 @@ struct llama_model_glm5_next : public llama_model_base {
};
+ // the NextN / MTP draft head: one DSA block appended after the trunk
+ struct graph_mtp : public graph {
+ graph_mtp(const llama_model & model, const llm_graph_params & params);
+ };
+
std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
};