Commit 41abbfd59 for llama.cpp
commit 41abbfd599fbdd3470fcae0a1fb6530ad8403cd7
Author: Aman Gupta <amangupta052@gmail.com>
Date: Mon Sep 14 22:16:30 2026 +0800
qwen4exp: enable rms_norm + mul fusion (#28896)
* qwen4exp: enable rms_norm + mul fusion
* use TENSOR_ALLOW_RESHAPE
diff --git a/src/models/qwen4exp.cpp b/src/models/qwen4exp.cpp
index 8ace95f73..e58d35034 100644
--- a/src/models/qwen4exp.cpp
+++ b/src/models/qwen4exp.cpp
@@ -157,7 +157,8 @@ void llama_model_qwen4exp::load_arch_tensors(llama_model_loader & ml) {
tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), { n_embd, n_vocab }, 0);
// there is no output_norm: the final hyper-connection mixer carries it
- hc_head_norm = create_tensor(tn(LLM_TENSOR_HC_HEAD_NORM, "weight"), { hc_dim }, 0);
+ // the gammas load as [n_embd, hc] so the grouped norm multiplies them without a graph reshape
+ hc_head_norm = create_tensor(tn(LLM_TENSOR_HC_HEAD_NORM, "weight"), { n_embd, hc }, TENSOR_ALLOW_RESHAPE);
hc_head_down = create_tensor(tn(LLM_TENSOR_HC_HEAD_DOWN, "weight"), { hc_dim, hc_lr }, 0);
hc_head_up = create_tensor(tn(LLM_TENSOR_HC_HEAD_UP, "weight"), { hc_lr, hc_dim }, 0);
@@ -203,11 +204,11 @@ void llama_model_qwen4exp::load_arch_tensors(llama_model_loader & ml) {
const int64_t conv_dim = key_dim * 2 + value_dim;
// two HC modules per layer: before the token mixer, before the MoE
- layer.hc_attn_norm = create_tensor(tn(LLM_TENSOR_HC_ATTN_NORM, "weight", il), { hc_dim }, 0);
+ layer.hc_attn_norm = create_tensor(tn(LLM_TENSOR_HC_ATTN_NORM, "weight", il), { n_embd, hc }, TENSOR_ALLOW_RESHAPE);
layer.hc_attn_down = create_tensor(tn(LLM_TENSOR_HC_ATTN_DOWN, "weight", il), { hc_dim, hc_lr }, 0);
layer.hc_attn_up = create_tensor(tn(LLM_TENSOR_HC_ATTN_UP, "weight", il), { hc_lr, hc_dim }, 0);
layer.hc_attn_inject = create_tensor(tn(LLM_TENSOR_HC_ATTN_INJECT, "weight", il), { hc_dim, hc }, 0);
- layer.hc_ffn_norm = create_tensor(tn(LLM_TENSOR_HC_FFN_NORM, "weight", il), { hc_dim }, 0);
+ layer.hc_ffn_norm = create_tensor(tn(LLM_TENSOR_HC_FFN_NORM, "weight", il), { n_embd, hc }, TENSOR_ALLOW_RESHAPE);
layer.hc_ffn_down = create_tensor(tn(LLM_TENSOR_HC_FFN_DOWN, "weight", il), { hc_dim, hc_lr }, 0);
layer.hc_ffn_up = create_tensor(tn(LLM_TENSOR_HC_FFN_UP, "weight", il), { hc_lr, hc_dim }, 0);
layer.hc_ffn_inject = create_tensor(tn(LLM_TENSOR_HC_FFN_INJECT, "weight", il), { hc_dim, hc }, 0);
@@ -240,9 +241,9 @@ void llama_model_qwen4exp::load_arch_tensors(llama_model_loader & ml) {
if (hparams.is_ple(il)) {
layer.ple_key = create_tensor(tn(LLM_TENSOR_PLE_KEY, "weight", il), { n_embd, hc_dim }, 0);
layer.ple_value = create_tensor(tn(LLM_TENSOR_PLE_VALUE, "weight", il), { n_embd, n_embd }, 0);
- layer.ple_norm_key = create_tensor(tn(LLM_TENSOR_PLE_NORM_KEY, "weight", il), { hc_dim }, 0);
- layer.ple_norm_query = create_tensor(tn(LLM_TENSOR_PLE_NORM_QUERY, "weight", il), { hc_dim }, 0);
- layer.ple_norm_conv = create_tensor(tn(LLM_TENSOR_PLE_NORM_CONV, "weight", il), { hc_dim }, 0);
+ layer.ple_norm_key = create_tensor(tn(LLM_TENSOR_PLE_NORM_KEY, "weight", il), { n_embd, hc }, TENSOR_ALLOW_RESHAPE);
+ layer.ple_norm_query = create_tensor(tn(LLM_TENSOR_PLE_NORM_QUERY, "weight", il), { n_embd, hc }, TENSOR_ALLOW_RESHAPE);
+ layer.ple_norm_conv = create_tensor(tn(LLM_TENSOR_PLE_NORM_CONV, "weight", il), { n_embd, hc }, TENSOR_ALLOW_RESHAPE);
layer.ple_conv1d = create_tensor(tn(LLM_TENSOR_PLE_CONV1D, "weight", il), { hparams.ple_conv_kernel, hc_dim }, 0);
}
@@ -275,11 +276,10 @@ ggml_tensor * llama_model_qwen4exp::graph::build_hc_mix(
const int64_t hc_dim = hc * n_embd;
const int64_t nt = x->ne[2];
- // grouped RMSNorm: reduce over one stream, then scale all streams with the [hc_dim] gamma
+ // grouped RMSNorm: reduce over one stream, then scale all streams with the [n_embd, hc] gamma
// the converter folded each gamma to (1 + w)
- ggml_tensor * xn = ggml_rms_norm(ctx0, x, hparams.f_norm_rms_eps);
+ ggml_tensor * xn = ggml_mul(ctx0, ggml_rms_norm(ctx0, x, hparams.f_norm_rms_eps), w_norm);
xn = ggml_reshape_2d(ctx0, xn, hc_dim, nt);
- xn = ggml_mul(ctx0, xn, w_norm);
cb(xn, "hc_norm", il);
ggml_tensor * lo = build_lora_mm(w_down, xn);
@@ -1200,13 +1200,10 @@ ggml_tensor * llama_model_qwen4exp::graph::build_ple(
ggml_tensor * key = build_lora_mm(model.layers[il].ple_key, emb);
ggml_tensor * value = build_lora_mm(model.layers[il].ple_value, emb);
- // both norms group over one hc stream, with a weight over the whole hc*n_embd layout
+ // both norms group over one hc stream, with a [n_embd, hc] weight
auto grouped_norm = [&](ggml_tensor * x, ggml_tensor * w) {
ggml_tensor * t = ggml_reshape_3d(ctx0, x, n_embd, hc, n_tokens);
- t = ggml_rms_norm(ctx0, t, hparams.f_norm_rms_eps);
- t = ggml_reshape_2d(ctx0, t, hc_dim, n_tokens);
- t = ggml_mul(ctx0, t, w);
- return ggml_reshape_3d(ctx0, t, n_embd, hc, n_tokens);
+ return ggml_mul(ctx0, ggml_rms_norm(ctx0, t, hparams.f_norm_rms_eps), w);
};
key = grouped_norm(key, model.layers[il].ple_norm_key);