kingjones777's picture
fix: complete MTP patch + QSA checkpoint fix (2026-09-17)
21a3548 verified
Raw History Blame Contribute Delete
31.7 kB
diff --git a/src/llama-model.cpp b/src/llama-model.cpp
index 7aca081..ee40fd1 100644
--- a/src/llama-model.cpp
+++ b/src/llama-model.cpp
@@ -1629,7 +1629,8 @@ const float * llama_model::tensor_split() const {
}
uint32_t llama_model::n_embd_pre_norm() const {
- return arch == LLM_ARCH_DEEPSEEK4 ? hparams.n_embd * hparams.n_hc : hparams.n_embd;
+ // hyper-connection archs: the pre-norm hidden is the wide [n_hc, n_embd] residual
+ return (arch == LLM_ARCH_DEEPSEEK4 || arch == LLM_ARCH_QWEN4EXP) ? hparams.n_embd * hparams.n_hc : hparams.n_embd;
}
uint32_t llama_model::n_gpu_layers() const {
@@ -2038,7 +2039,8 @@ llama_memory_i * llama_model::create_memory(const llama_memory_params & params,
// attention KV cache for the MTP context instead of the hybrid wrapper.
const bool mtp_on_hybrid_qwen35 =
params.ctx_type == LLAMA_CONTEXT_TYPE_MTP &&
- (arch == LLM_ARCH_QWEN35 || arch == LLM_ARCH_QWEN35MOE || arch == LLM_ARCH_BAILINGMOE3);
+ (arch == LLM_ARCH_QWEN35 || arch == LLM_ARCH_QWEN35MOE || arch == LLM_ARCH_BAILINGMOE3 ||
+ arch == LLM_ARCH_QWEN4EXP);
const bool step35_with_mtp =
arch == LLM_ARCH_STEP35 && hparams.nextn_predict_layers > 0;
const bool mtp_on_step35 =
diff --git a/src/models/models.h b/src/models/models.h
index aea1cb1..2db61a0 100644
--- a/src/models/models.h
+++ b/src/models/models.h
@@ -1947,9 +1947,11 @@ struct llama_model_qwen4exp : public llama_model_base {
void load_arch_hparams(llama_model_loader & ml) override;
void load_arch_tensors(llama_model_loader & ml) override;
- struct graph : public llm_build_delta_net_base {
- graph(const llama_model & model, const llm_graph_params & params);
- private:
+ // shared block builders: the trunk forward and the MTP draft forward run the
+ // same hc/attention/MoE block, so both graphs derive from this
+ struct graph_base : public llm_build_delta_net_base {
+ graph_base(const llama_model & model, const llm_graph_params & params);
+
// HC replaces every layer norm: residual is [n_embd, hc, n_tokens]
ggml_tensor * build_hc_mix(
ggml_tensor * x,
@@ -1960,12 +1962,18 @@ struct llama_model_qwen4exp : public llama_model_base {
ggml_tensor ** inject,
int il);
+ // collapse the hc parallel streams [n_embd, hc, T] by their mean into [n_embd, T]
+ ggml_tensor * build_hc_collapse(
+ ggml_tensor * x,
+ int il);
+
ggml_tensor * build_hc_combine(
ggml_tensor * residual,
ggml_tensor * block_out,
ggml_tensor * inject,
int il);
+ // mctx_hyb null means a plain attention context with no indexer cache: dense
ggml_tensor * build_layer_attn(
llm_graph_input_attn_kv * inp_attn,
const llama_memory_hybrid_idx_context * mctx_hyb,
@@ -1994,12 +2002,18 @@ struct llama_model_qwen4exp : public llama_model_base {
int * sections,
int il);
- ggml_tensor * build_layer_attn_linear(
- llm_graph_input_rs * inp,
+ ggml_tensor * build_layer_ffn(
ggml_tensor * cur,
int il);
- ggml_tensor * build_layer_ffn(
+ const llama_model & model;
+ };
+
+ struct graph : public graph_base {
+ graph(const llama_model & model, const llm_graph_params & params);
+ private:
+ ggml_tensor * build_layer_attn_linear(
+ llm_graph_input_rs * inp,
ggml_tensor * cur,
int il);
@@ -2032,8 +2046,12 @@ struct llama_model_qwen4exp : public llama_model_base {
std::pair<ggml_tensor *, ggml_tensor *> build_qkvz(
ggml_tensor * input,
int il);
+ };
- const llama_model & model;
+ // LLM_GRAPH_TYPE_DECODER_MTP draft head: the trailing nextn block reading the
+ // target's wide pre-norm hidden rows
+ struct graph_mtp : public graph_base {
+ graph_mtp(const llama_model & model, const llm_graph_params & params);
};
std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
diff --git a/src/models/qwen4exp.cpp b/src/models/qwen4exp.cpp
index ca3064d..2dd6d4f 100644
--- a/src/models/qwen4exp.cpp
+++ b/src/models/qwen4exp.cpp
@@ -9,6 +9,9 @@ void llama_model_qwen4exp::load_arch_hparams(llama_model_loader & ml) {
ml.get_key_or_arr(LLM_KV_ROPE_DIMENSION_SECTIONS, hparams.rope_sections, 4, true);
+ // trailing nextn (MTP) blocks; absent in trunk-only files
+ ml.get_key(LLM_KV_NEXTN_PREDICT_LAYERS, hparams.nextn_predict_layers, false);
+
ml.get_key(LLM_KV_SSM_CONV_KERNEL, hparams.ssm_d_conv);
ml.get_key(LLM_KV_SSM_INNER_SIZE, hparams.ssm_d_inner);
@@ -114,7 +117,13 @@ void llama_model_qwen4exp::load_arch_tensors(llama_model_loader & ml) {
{ hparams.ple_head_dim, ple_rows }, 0);
}
- for (int il = 0; il < n_layer; ++il) {
+ // MTP sidecars carry only the trailing nextn block: relax the trunk tensors then
+ const uint32_t n_main = n_layer - hparams.nextn_predict_layers;
+ const bool mtp_only = (hparams.nextn_predict_layers > 0) &&
+ (ml.get_weight("blk.0.hc_attn_norm.weight") == nullptr);
+ const int trunk_flags = mtp_only ? TENSOR_NOT_REQUIRED : 0;
+
+ auto load_block_trunk = [&](int il, int flags) {
auto & layer = layers[il];
const int64_t n_ff_exp = hparams.n_ff_exp ? hparams.n_ff_exp : n_ff / n_expert_used;
@@ -129,50 +138,94 @@ void llama_model_qwen4exp::load_arch_tensors(llama_model_loader & ml) {
const int64_t conv_dim = key_dim * 2 + value_dim;
// two HC modules per layer: before the token mixer, before the MoE
- layer.hc_attn_norm = create_tensor(tn(LLM_TENSOR_HC_ATTN_NORM, "weight", il), { hc_dim }, 0);
- layer.hc_attn_down = create_tensor(tn(LLM_TENSOR_HC_ATTN_DOWN, "weight", il), { hc_dim, hc_lr }, 0);
- layer.hc_attn_up = create_tensor(tn(LLM_TENSOR_HC_ATTN_UP, "weight", il), { hc_lr, hc_dim }, 0);
- layer.hc_attn_inject = create_tensor(tn(LLM_TENSOR_HC_ATTN_INJECT, "weight", il), { hc_dim, hc }, 0);
- layer.hc_ffn_norm = create_tensor(tn(LLM_TENSOR_HC_FFN_NORM, "weight", il), { hc_dim }, 0);
- layer.hc_ffn_down = create_tensor(tn(LLM_TENSOR_HC_FFN_DOWN, "weight", il), { hc_dim, hc_lr }, 0);
- layer.hc_ffn_up = create_tensor(tn(LLM_TENSOR_HC_FFN_UP, "weight", il), { hc_lr, hc_dim }, 0);
- layer.hc_ffn_inject = create_tensor(tn(LLM_TENSOR_HC_FFN_INJECT, "weight", il), { hc_dim, hc }, 0);
+ layer.hc_attn_norm = create_tensor(tn(LLM_TENSOR_HC_ATTN_NORM, "weight", il), { hc_dim }, flags);
+ layer.hc_attn_down = create_tensor(tn(LLM_TENSOR_HC_ATTN_DOWN, "weight", il), { hc_dim, hc_lr }, flags);
+ layer.hc_attn_up = create_tensor(tn(LLM_TENSOR_HC_ATTN_UP, "weight", il), { hc_lr, hc_dim }, flags);
+ layer.hc_attn_inject = create_tensor(tn(LLM_TENSOR_HC_ATTN_INJECT, "weight", il), { hc_dim, hc }, flags);
+ layer.hc_ffn_norm = create_tensor(tn(LLM_TENSOR_HC_FFN_NORM, "weight", il), { hc_dim }, flags);
+ layer.hc_ffn_down = create_tensor(tn(LLM_TENSOR_HC_FFN_DOWN, "weight", il), { hc_dim, hc_lr }, flags);
+ layer.hc_ffn_up = create_tensor(tn(LLM_TENSOR_HC_FFN_UP, "weight", il), { hc_lr, hc_dim }, flags);
+ layer.hc_ffn_inject = create_tensor(tn(LLM_TENSOR_HC_FFN_INJECT, "weight", il), { hc_dim, hc }, flags);
if (!hparams.is_recurrent(il)) {
// full attention: wq holds [q|gate] interleaved per head
- create_tensor_qkv(layer, il, n_embd, n_embd_head_k * n_head * 2, n_embd_k_gqa, n_embd_v_gqa, 0);
- layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", il), { n_embd_head_k * n_head, n_embd }, 0);
+ create_tensor_qkv(layer, il, n_embd, n_embd_head_k * n_head * 2, n_embd_k_gqa, n_embd_v_gqa, flags);
+ layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", il), { n_embd_head_k * n_head, n_embd }, flags);
- layer.attn_q_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_NORM, "weight", il), { n_embd_head_k }, 0);
- layer.attn_k_norm = create_tensor(tn(LLM_TENSOR_ATTN_K_NORM, "weight", il), { n_embd_head_k }, 0);
+ layer.attn_q_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_NORM, "weight", il), { n_embd_head_k }, flags);
+ layer.attn_k_norm = create_tensor(tn(LLM_TENSOR_ATTN_K_NORM, "weight", il), { n_embd_head_k }, flags);
const int64_t idx_dim = hparams.indexer_head_size;
- layer.index_q_proj = create_tensor(tn(LLM_TENSOR_INDEXER_Q_PROJ, "weight", il), { n_embd, hparams.indexer_n_head * idx_dim }, 0);
- layer.index_k_proj = create_tensor(tn(LLM_TENSOR_INDEXER_K_PROJ, "weight", il), { n_embd, idx_dim }, 0);
- layer.index_q_norm = create_tensor(tn(LLM_TENSOR_INDEXER_Q_NORM, "weight", il), { idx_dim }, 0);
- layer.index_k_norm = create_tensor(tn(LLM_TENSOR_INDEXER_K_NORM, "weight", il), { idx_dim }, 0);
+ layer.index_q_proj = create_tensor(tn(LLM_TENSOR_INDEXER_Q_PROJ, "weight", il), { n_embd, hparams.indexer_n_head * idx_dim }, flags);
+ layer.index_k_proj = create_tensor(tn(LLM_TENSOR_INDEXER_K_PROJ, "weight", il), { n_embd, idx_dim }, flags);
+ layer.index_q_norm = create_tensor(tn(LLM_TENSOR_INDEXER_Q_NORM, "weight", il), { idx_dim }, flags);
+ layer.index_k_norm = create_tensor(tn(LLM_TENSOR_INDEXER_K_NORM, "weight", il), { idx_dim }, flags);
} else {
- layer.wqkv = create_tensor(tn(LLM_TENSOR_ATTN_QKV, "weight", il), { n_embd, key_dim * 2 + value_dim }, 0);
- layer.wqkv_gate = create_tensor(tn(LLM_TENSOR_ATTN_GATE, "weight", il), { n_embd, value_dim }, 0);
- layer.ssm_conv1d = create_tensor(tn(LLM_TENSOR_SSM_CONV1D, "weight", il), { hparams.ssm_d_conv, conv_dim }, 0);
- layer.ssm_dt = create_tensor(tn(LLM_TENSOR_SSM_DT, "bias", il), { hparams.ssm_dt_rank }, 0);
- layer.ssm_a = create_tensor(tn(LLM_TENSOR_SSM_A_NOSCAN, il), { hparams.ssm_dt_rank }, 0);
- layer.ssm_beta = create_tensor(tn(LLM_TENSOR_SSM_BETA, "weight", il), { n_embd, n_v_heads }, 0);
- layer.ssm_alpha = create_tensor(tn(LLM_TENSOR_SSM_ALPHA, "weight", il), { n_embd, n_v_heads }, 0);
- layer.ssm_norm = create_tensor(tn(LLM_TENSOR_SSM_NORM, "weight", il), { head_v_dim }, 0);
- layer.ssm_out = create_tensor(tn(LLM_TENSOR_SSM_OUT, "weight", il), { value_dim, n_embd }, 0);
+ layer.wqkv = create_tensor(tn(LLM_TENSOR_ATTN_QKV, "weight", il), { n_embd, key_dim * 2 + value_dim }, flags);
+ layer.wqkv_gate = create_tensor(tn(LLM_TENSOR_ATTN_GATE, "weight", il), { n_embd, value_dim }, flags);
+ layer.ssm_conv1d = create_tensor(tn(LLM_TENSOR_SSM_CONV1D, "weight", il), { hparams.ssm_d_conv, conv_dim }, flags);
+ layer.ssm_dt = create_tensor(tn(LLM_TENSOR_SSM_DT, "bias", il), { hparams.ssm_dt_rank }, flags);
+ layer.ssm_a = create_tensor(tn(LLM_TENSOR_SSM_A_NOSCAN, il), { hparams.ssm_dt_rank }, flags);
+ layer.ssm_beta = create_tensor(tn(LLM_TENSOR_SSM_BETA, "weight", il), { n_embd, n_v_heads }, flags);
+ layer.ssm_alpha = create_tensor(tn(LLM_TENSOR_SSM_ALPHA, "weight", il), { n_embd, n_v_heads }, flags);
+ layer.ssm_norm = create_tensor(tn(LLM_TENSOR_SSM_NORM, "weight", il), { head_v_dim }, flags);
+ layer.ssm_out = create_tensor(tn(LLM_TENSOR_SSM_OUT, "weight", il), { value_dim, n_embd }, flags);
}
if (hparams.is_ple(il)) {
- layer.ple_key = create_tensor(tn(LLM_TENSOR_PLE_KEY, "weight", il), { n_embd, hc_dim }, 0);
- layer.ple_value = create_tensor(tn(LLM_TENSOR_PLE_VALUE, "weight", il), { n_embd, n_embd }, 0);
- layer.ple_norm_key = create_tensor(tn(LLM_TENSOR_PLE_NORM_KEY, "weight", il), { hc_dim }, 0);
- layer.ple_norm_query = create_tensor(tn(LLM_TENSOR_PLE_NORM_QUERY, "weight", il), { hc_dim }, 0);
- layer.ple_norm_conv = create_tensor(tn(LLM_TENSOR_PLE_NORM_CONV, "weight", il), { hc_dim }, 0);
- layer.ple_conv1d = create_tensor(tn(LLM_TENSOR_PLE_CONV1D, "weight", il), { hparams.ple_conv_kernel, hc_dim }, 0);
+ layer.ple_key = create_tensor(tn(LLM_TENSOR_PLE_KEY, "weight", il), { n_embd, hc_dim }, flags);
+ layer.ple_value = create_tensor(tn(LLM_TENSOR_PLE_VALUE, "weight", il), { n_embd, n_embd }, flags);
+ layer.ple_norm_key = create_tensor(tn(LLM_TENSOR_PLE_NORM_KEY, "weight", il), { hc_dim }, flags);
+ layer.ple_norm_query = create_tensor(tn(LLM_TENSOR_PLE_NORM_QUERY, "weight", il), { hc_dim }, flags);
+ layer.ple_norm_conv = create_tensor(tn(LLM_TENSOR_PLE_NORM_CONV, "weight", il), { hc_dim }, flags);
+ layer.ple_conv1d = create_tensor(tn(LLM_TENSOR_PLE_CONV1D, "weight", il), { hparams.ple_conv_kernel, hc_dim }, flags);
}
+ layer.ffn_gate_inp = create_tensor(tn(LLM_TENSOR_FFN_GATE_INP, "weight", il), { n_embd, n_expert }, flags);
+ layer.ffn_down_exps = create_tensor(tn(LLM_TENSOR_FFN_DOWN_EXPS, "weight", il), { n_ff_exp, n_embd, n_expert }, flags);
+ create_tensor_gate_up_exps(layer, il, n_embd, n_ff_exp, n_expert, flags);
+
+ layer.ffn_gate_inp_shexp = create_tensor(tn(LLM_TENSOR_FFN_GATE_INP_SHEXP, "weight", il), { n_embd }, flags);
+ layer.ffn_gate_shexp = create_tensor(tn(LLM_TENSOR_FFN_GATE_SHEXP, "weight", il), { n_embd, n_ff_shexp }, flags);
+ layer.ffn_up_shexp = create_tensor(tn(LLM_TENSOR_FFN_UP_SHEXP, "weight", il), { n_embd, n_ff_shexp }, flags);
+ layer.ffn_down_shexp = create_tensor(tn(LLM_TENSOR_FFN_DOWN_SHEXP, "weight", il), { n_ff_shexp, n_embd }, flags);
+ };
+
+ // the trailing nextn block: a full-attention decoder layer carrying the whole
+ // hyper-connection + MoE kit, plus the three nextn projection/norm tensors
+ auto load_block_mtp = [&](int il) {
+ auto & layer = layers[il];
+
+ GGML_ASSERT(!hparams.is_recurrent(il) && "the qwen4exp MTP block must be a full-attention layer");
+ GGML_ASSERT(!hparams.is_ple(il) && "the qwen4exp MTP block carries no PLE tensors");
+
+ const int64_t n_ff_exp = hparams.n_ff_exp ? hparams.n_ff_exp : n_ff / n_expert_used;
+ const int64_t n_ff_shexp = hparams.n_ff_shexp ? hparams.n_ff_shexp : n_ff;
+
+ layer.hc_attn_norm = create_tensor(tn(LLM_TENSOR_HC_ATTN_NORM, "weight", il), { hc_dim }, 0);
+ layer.hc_attn_down = create_tensor(tn(LLM_TENSOR_HC_ATTN_DOWN, "weight", il), { hc_dim, hc_lr }, 0);
+ layer.hc_attn_up = create_tensor(tn(LLM_TENSOR_HC_ATTN_UP, "weight", il), { hc_lr, hc_dim }, 0);
+ layer.hc_attn_inject = create_tensor(tn(LLM_TENSOR_HC_ATTN_INJECT, "weight", il), { hc_dim, hc }, 0);
+ layer.hc_ffn_norm = create_tensor(tn(LLM_TENSOR_HC_FFN_NORM, "weight", il), { hc_dim }, 0);
+ layer.hc_ffn_down = create_tensor(tn(LLM_TENSOR_HC_FFN_DOWN, "weight", il), { hc_dim, hc_lr }, 0);
+ layer.hc_ffn_up = create_tensor(tn(LLM_TENSOR_HC_FFN_UP, "weight", il), { hc_lr, hc_dim }, 0);
+ layer.hc_ffn_inject = create_tensor(tn(LLM_TENSOR_HC_FFN_INJECT, "weight", il), { hc_dim, hc }, 0);
+
+ create_tensor_qkv(layer, il, n_embd, n_embd_head_k * n_head * 2, n_embd_k_gqa, n_embd_v_gqa, 0);
+ layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", il), { n_embd_head_k * n_head, n_embd }, 0);
+
+ layer.attn_q_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_NORM, "weight", il), { n_embd_head_k }, 0);
+ layer.attn_k_norm = create_tensor(tn(LLM_TENSOR_ATTN_K_NORM, "weight", il), { n_embd_head_k }, 0);
+
+ // present in the sidecar even though the MTP block's compress ratio is 0
+ // (dense attention): load them so the file carries no unused tensors
+ const int64_t idx_dim = hparams.indexer_head_size;
+ layer.index_q_proj = create_tensor(tn(LLM_TENSOR_INDEXER_Q_PROJ, "weight", il), { n_embd, hparams.indexer_n_head * idx_dim }, 0);
+ layer.index_k_proj = create_tensor(tn(LLM_TENSOR_INDEXER_K_PROJ, "weight", il), { n_embd, idx_dim }, 0);
+ layer.index_q_norm = create_tensor(tn(LLM_TENSOR_INDEXER_Q_NORM, "weight", il), { idx_dim }, 0);
+ layer.index_k_norm = create_tensor(tn(LLM_TENSOR_INDEXER_K_NORM, "weight", il), { idx_dim }, 0);
+
layer.ffn_gate_inp = create_tensor(tn(LLM_TENSOR_FFN_GATE_INP, "weight", il), { n_embd, n_expert }, 0);
layer.ffn_down_exps = create_tensor(tn(LLM_TENSOR_FFN_DOWN_EXPS, "weight", il), { n_ff_exp, n_embd, n_expert }, 0);
create_tensor_gate_up_exps(layer, il, n_embd, n_ff_exp, n_expert, 0);
@@ -181,13 +234,39 @@ void llama_model_qwen4exp::load_arch_tensors(llama_model_loader & ml) {
layer.ffn_gate_shexp = create_tensor(tn(LLM_TENSOR_FFN_GATE_SHEXP, "weight", il), { n_embd, n_ff_shexp }, 0);
layer.ffn_up_shexp = create_tensor(tn(LLM_TENSOR_FFN_UP_SHEXP, "weight", il), { n_embd, n_ff_shexp }, 0);
layer.ffn_down_shexp = create_tensor(tn(LLM_TENSOR_FFN_DOWN_SHEXP, "weight", il), { n_ff_shexp, n_embd }, 0);
+
+ // nextn-specific tensors. hnorm spans the whole hc layout, not n_embd
+ layer.nextn.eh_proj = create_tensor(tn(LLM_TENSOR_NEXTN_EH_PROJ, "weight", il), { 2 * n_embd, n_embd }, 0);
+ layer.nextn.enorm = create_tensor(tn(LLM_TENSOR_NEXTN_ENORM, "weight", il), { n_embd }, 0);
+ layer.nextn.hnorm = create_tensor(tn(LLM_TENSOR_NEXTN_HNORM, "weight", il), { hc_dim }, 0);
+
+ // optional shared-head tensors present in some sidecar heads (the
+ // 35-tensor Q4_K_M/Q6_K/ROCmFP4 heads carry shared_head_norm); load
+ // them so the file carries no unconsumed tensors. shared_head_norm
+ // spans the hc layout like hnorm, not n_embd.
+ layer.nextn.embed_tokens = create_tensor(tn(LLM_TENSOR_NEXTN_EMBED_TOKENS, "weight", il), { n_embd, n_vocab }, TENSOR_NOT_REQUIRED);
+ layer.nextn.shared_head_head = create_tensor(tn(LLM_TENSOR_NEXTN_SHARED_HEAD_HEAD, "weight", il), { n_embd, n_vocab }, TENSOR_NOT_REQUIRED);
+ layer.nextn.shared_head_norm = create_tensor(tn(LLM_TENSOR_NEXTN_SHARED_HEAD_NORM, "weight", il), { hc_dim }, TENSOR_NOT_REQUIRED);
+ };
+
+ for (int i = 0; i < (int) n_main; ++i) {
+ load_block_trunk(i, trunk_flags);
+ }
+ for (int i = (int) n_main; i < n_layer; ++i) {
+ load_block_mtp(i);
}
}
std::unique_ptr<llm_graph_context> llama_model_qwen4exp::build_arch_graph(const llm_graph_params & params) const {
+ if (params.gtype == LLM_GRAPH_TYPE_DECODER_MTP) {
+ return std::make_unique<graph_mtp>(*this, params);
+ }
return std::make_unique<graph>(*this, params);
}
+llama_model_qwen4exp::graph_base::graph_base(const llama_model & model, const llm_graph_params & params) :
+ llm_build_delta_net_base(params), model(model) {}
+
// Hyper-connections replace every layer norm: the state between blocks is `hc`
// parallel residual streams [n_embd, hc, T]; each block reads one mixed [n_embd, T]
// view and writes back through per-stream injection weights.
@@ -195,7 +274,7 @@ std::unique_ptr<llm_graph_context> llama_model_qwen4exp::build_arch_graph(const
// low-rank down/silu/up gate with a plain mean collapse.
// The mix output is [n_embd, T]; `inject` receives the [hc, T] scatter weights.
-ggml_tensor * llama_model_qwen4exp::graph::build_hc_mix(
+ggml_tensor * llama_model_qwen4exp::graph_base::build_hc_mix(
ggml_tensor * x,
ggml_tensor * w_norm,
ggml_tensor * w_down,
@@ -222,28 +301,39 @@ ggml_tensor * llama_model_qwen4exp::graph::build_hc_mix(
ggml_tensor * gated = ggml_mul(ctx0, xn, gate);
gated = ggml_reshape_3d(ctx0, gated, n_embd, hc, nt);
+ ggml_tensor * mixed = build_hc_collapse(gated, il);
+
+ if (inject) {
+ *inject = build_lora_mm(w_inject, xn);
+ cb(*inject, "hc_inject", il);
+ }
+
+ return mixed;
+}
+
+ggml_tensor * llama_model_qwen4exp::graph_base::build_hc_collapse(
+ ggml_tensor * x,
+ int il) {
+ const int64_t hc = hparams.n_hc;
+ const int64_t nt = x->ne[2];
+
// collapse the streams by their mean
- ggml_tensor * mixed = ggml_view_2d(ctx0, gated, n_embd, nt,
- ggml_row_size(gated->type, n_embd) * hc, 0);
+ ggml_tensor * mixed = ggml_view_2d(ctx0, x, n_embd, nt,
+ ggml_row_size(x->type, n_embd) * hc, 0);
mixed = ggml_cont(ctx0, mixed);
for (int64_t c = 1; c < hc; ++c) {
- ggml_tensor * s = ggml_view_2d(ctx0, gated, n_embd, nt,
- ggml_row_size(gated->type, n_embd) * hc,
- ggml_row_size(gated->type, n_embd) * c);
+ ggml_tensor * s = ggml_view_2d(ctx0, x, n_embd, nt,
+ ggml_row_size(x->type, n_embd) * hc,
+ ggml_row_size(x->type, n_embd) * c);
mixed = ggml_add(ctx0, mixed, s);
}
mixed = ggml_scale(ctx0, mixed, 1.0f / (float) hc);
cb(mixed, "hc_mixed", il);
- if (inject) {
- *inject = build_lora_mm(w_inject, xn);
- cb(*inject, "hc_inject", il);
- }
-
return mixed;
}
-ggml_tensor * llama_model_qwen4exp::graph::build_hc_combine(
+ggml_tensor * llama_model_qwen4exp::graph_base::build_hc_combine(
ggml_tensor * residual,
ggml_tensor * block_out,
ggml_tensor * inject,
@@ -267,7 +357,7 @@ ggml_tensor * llama_model_qwen4exp::graph::build_hc_combine(
}
llama_model_qwen4exp::graph::graph(const llama_model & model, const llm_graph_params & params) :
- llm_build_delta_net_base(params), model(model) {
+ graph_base(model, params) {
const int64_t hc = hparams.n_hc;
GGML_ASSERT(hparams.n_embd_head_v() == hparams.n_embd_head_k());
@@ -344,6 +434,13 @@ llama_model_qwen4exp::graph::graph(const llama_model & model, const llm_graph_pa
cb(res_hc, "l_last", il);
}
+ // the wide residual before the final mixer is what an MTP draft consumes
+ // (hnorm spans the whole hc layout): export it for all tokens, like deepseek4
+ if (cparams.embeddings_pre_norm) {
+ res->t_h_pre_norm = res_hc;
+ cb(res->t_h_pre_norm, "h_pre_norm", -1);
+ }
+
// the final mixer is the output norm: there is no separate one
ggml_tensor * cur = build_hc_mix(res_hc,
model.hc_head_norm, model.hc_head_down, model.hc_head_up,
@@ -363,6 +460,132 @@ llama_model_qwen4exp::graph::graph(const llama_model & model, const llm_graph_pa
ggml_build_forward_expand(gf, cur);
}
+// LLM_GRAPH_TYPE_DECODER_MTP draft head. One full-attention nextn block:
+// hnorm(h) and enorm(embed(t)) collapse to [n_embd], concat, eh_proj,
+// then the same hc/attention/MoE block the trunk runs, reusing its builders.
+// The draft context is a plain KV cache (mtp_on_hybrid_qwen35), so the attention
+// is dense: the nextn block's compress ratio is 0 and there is no indexer cache.
+llama_model_qwen4exp::graph_mtp::graph_mtp(const llama_model & model, const llm_graph_params & params)
+ : graph_base(model, params) {
+ GGML_ASSERT(hparams.nextn_predict_layers > 0 && "qwen4exp MTP requires nextn_predict_layers > 0");
+ GGML_ASSERT(hparams.nextn_predict_layers == 1 && "qwen4exp MTP currently only supports a single MTP block");
+
+ const int64_t hc = hparams.n_hc;
+ const int64_t hc_dim = hc * n_embd;
+
+ // the MTP block lives at the source file's original layer index
+ const int il = (int) hparams.n_layer - (int) hparams.nextn_predict_layers;
+ const auto & layer = model.layers[il];
+
+ GGML_ASSERT(layer.nextn.eh_proj && "qwen4exp MTP block missing nextn.eh_proj");
+ GGML_ASSERT(layer.nextn.enorm && "qwen4exp MTP block missing nextn.enorm");
+ GGML_ASSERT(layer.nextn.hnorm && "qwen4exp MTP block missing nextn.hnorm");
+ GGML_ASSERT(!hparams.is_recurrent(il) && "the qwen4exp MTP block must be a full-attention layer");
+ GGML_ASSERT(!hparams.is_ple(il) && "the qwen4exp MTP block carries no PLE tensors");
+
+ int sections[4];
+ std::copy(std::begin(hparams.rope_sections), std::begin(hparams.rope_sections) + 4, sections);
+
+ // inputs: the accepted/next token ids plus the target's wide pre-norm hidden rows
+ auto inp = std::make_unique<llm_graph_input_embd_h>(hc_dim);
+
+ inp->tokens = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_tokens);
+ ggml_set_input(inp->tokens);
+
+ inp->embd = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, hc_dim, n_tokens);
+ ggml_set_input(inp->embd);
+
+ inp->h = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, hc_dim, n_tokens);
+ ggml_set_input(inp->h);
+ ggml_set_name(inp->h, "mtp_h_input");
+
+ // raw views on the inputs; the unique_ptr itself moves below
+ ggml_tensor * inp_tokens = inp->tokens;
+ ggml_tensor * h_input = inp->h;
+
+ res->add_input(std::move(inp));
+
+ ggml_tensor * inp_pos = build_inp_pos();
+ ggml_tensor * inp_out_ids = build_inp_out_ids();
+ auto * inp_attn = build_attn_inp_kv();
+
+ // e-side: embed the token, norm at n_embd width. The token embedding is shared by
+ // every stream, so broadcast it to hc copies.
+ ggml_tensor * e_cur = ggml_get_rows(ctx0, model.tok_embd, inp_tokens);
+ cb(e_cur, "mtp_tok_embd", il);
+ e_cur = build_norm(e_cur, layer.nextn.enorm, nullptr, LLM_NORM_RMS, il);
+ e_cur = ggml_repeat_4d(ctx0,
+ ggml_reshape_3d(ctx0, e_cur, n_embd, 1, n_tokens),
+ n_embd, hc, n_tokens, 1);
+ cb(e_cur, "mtp_enorm", il);
+
+ // h-side: group-norm each hc stream (hnorm spans the whole hc layout). The streams
+ // stay distinct: the combiner runs per hyper-connection stream on the wide hidden
+ // state, and pooling them first is what dropped the acceptance.
+ ggml_tensor * h_cur = ggml_reshape_3d(ctx0, h_input, n_embd, hc, n_tokens);
+ h_cur = ggml_rms_norm(ctx0, h_cur, hparams.f_norm_rms_eps);
+ h_cur = ggml_reshape_2d(ctx0, h_cur, hc_dim, n_tokens);
+ h_cur = ggml_mul(ctx0, h_cur, layer.nextn.hnorm);
+ h_cur = ggml_reshape_3d(ctx0, h_cur, n_embd, hc, n_tokens); // JAY-FIX: restore per-stream 3D before concat
+ cb(h_cur, "mtp_hnorm", il);
+
+ ggml_tensor * concat = ggml_concat(ctx0, e_cur, h_cur, /*dim=*/ 0);
+ cb(concat, "mtp_concat", il);
+
+ // [2*n_embd, hc, T] @ eh_proj [2*n_embd, n_embd]: one matmul per stream, i.e.
+ // fc_embedding @ e + fc_hidden @ h for each of the hc streams
+ ggml_tensor * cur = build_lora_mm(layer.nextn.eh_proj, concat);
+ cb(cur, "mtp_eh_proj", il);
+
+ // the projection is already the wide residual: hc distinct streams, not hc copies
+ ggml_tensor * res_hc = cur;
+ cb(res_hc, "mtp_hc_init", il);
+
+ // attention sublayer
+ ggml_tensor * inject = nullptr;
+ ggml_tensor * mixed = build_hc_mix(res_hc,
+ layer.hc_attn_norm, layer.hc_attn_down, layer.hc_attn_up, layer.hc_attn_inject,
+ &inject, il);
+
+ ggml_build_forward_expand(gf, mixed);
+
+ cur = build_layer_attn(inp_attn, nullptr, mixed, inp_pos, sections, il);
+
+ res_hc = build_hc_combine(res_hc, cur, inject, il);
+
+ // MoE sublayer
+ mixed = build_hc_mix(res_hc,
+ layer.hc_ffn_norm, layer.hc_ffn_down, layer.hc_ffn_up, layer.hc_ffn_inject,
+ &inject, il);
+
+ cur = build_layer_ffn(mixed, il);
+ cb(cur, "mtp_ffn_out", il);
+
+ res_hc = build_hc_combine(res_hc, cur, inject, il);
+
+ // the next draft step consumes the same wide pre-mix residual the target exports
+ res->t_h_pre_norm = res_hc;
+ cb(res_hc, "h_nextn", -1);
+
+ // the final mixer is the output norm: the sidecar carries its own hc head
+ cur = build_hc_mix(res_hc,
+ model.hc_head_norm, model.hc_head_down, model.hc_head_up,
+ nullptr, nullptr, -1);
+
+ if (inp_out_ids) {
+ cur = ggml_get_rows(ctx0, cur, inp_out_ids);
+ }
+
+ cb(cur, "result_norm", -1);
+ res->t_embd = cur;
+
+ cur = build_lora_mm(model.output, cur, model.output_s);
+ cb(cur, "result_output", -1);
+ res->t_logits = cur;
+
+ ggml_build_forward_expand(gf, cur);
+}
+
std::pair<ggml_tensor *, ggml_tensor *> llama_model_qwen4exp::graph::build_qkvz(
ggml_tensor * input,
int il) {
@@ -418,7 +641,7 @@ public:
const uint32_t ratio;
};
-ggml_tensor * llama_model_qwen4exp::graph::build_qsa_top_k(
+ggml_tensor * llama_model_qwen4exp::graph_base::build_qsa_top_k(
const llama_memory_hybrid_idx_context * mctx_hyb,
ggml_tensor * cur,
ggml_tensor * inp_pos,
@@ -541,7 +764,7 @@ ggml_tensor * llama_model_qwen4exp::graph::build_qsa_top_k(
// for llm_graph_input_attn_k_dsa, and the mask construction below is a copy of that one.
// It is kept here rather than factored into a shared helper so that the attention path used
// by every other architecture is untouched by this arch.
-ggml_tensor * llama_model_qwen4exp::graph::build_attn_qsa(
+ggml_tensor * llama_model_qwen4exp::graph_base::build_attn_qsa(
llm_graph_input_attn_kv * inp,
ggml_tensor * q_cur,
ggml_tensor * k_cur,
@@ -607,7 +830,7 @@ ggml_tensor * llama_model_qwen4exp::graph::build_attn_qsa(
return cur;
}
-ggml_tensor * llama_model_qwen4exp::graph::build_layer_attn(
+ggml_tensor * llama_model_qwen4exp::graph_base::build_layer_attn(
llm_graph_input_attn_kv * inp,
const llama_memory_hybrid_idx_context * mctx_hyb,
ggml_tensor * cur,
@@ -617,8 +840,9 @@ ggml_tensor * llama_model_qwen4exp::graph::build_layer_attn(
const int64_t n_embd_head = hparams.n_embd_head_v();
GGML_ASSERT(n_embd_head == hparams.n_embd_head_k());
- // indexer reads the same block input as q/k/v; no cache or no ratio means dense
- const bool qsa = mctx_hyb->get_idx() != nullptr && hparams.attn_compress_ratio[il] > 0;
+ // indexer reads the same block input as q/k/v; no cache or no ratio means dense.
+ // a null mctx_hyb is a plain attention context (the MTP draft): always dense
+ const bool qsa = mctx_hyb != nullptr && mctx_hyb->get_idx() != nullptr && hparams.attn_compress_ratio[il] > 0;
ggml_tensor * top_k = qsa ? build_qsa_top_k(mctx_hyb, cur, inp_pos, sections, il) : nullptr;
@@ -834,7 +1058,7 @@ ggml_tensor * llama_model_qwen4exp::graph::build_layer_attn_linear(
return cur;
}
-ggml_tensor * llama_model_qwen4exp::graph::build_layer_ffn(ggml_tensor * cur, const int il) {
+ggml_tensor * llama_model_qwen4exp::graph_base::build_layer_ffn(ggml_tensor * cur, const int il) {
// Check if this is an MoE layer
GGML_ASSERT(model.layers[il].ffn_gate_inp != nullptr);