diff --git a/src/llama-model.cpp b/src/llama-model.cpp index 7aca081..ee40fd1 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -1629,7 +1629,8 @@ const float * llama_model::tensor_split() const { } uint32_t llama_model::n_embd_pre_norm() const { - return arch == LLM_ARCH_DEEPSEEK4 ? hparams.n_embd * hparams.n_hc : hparams.n_embd; + // hyper-connection archs: the pre-norm hidden is the wide [n_hc, n_embd] residual + return (arch == LLM_ARCH_DEEPSEEK4 || arch == LLM_ARCH_QWEN4EXP) ? hparams.n_embd * hparams.n_hc : hparams.n_embd; } uint32_t llama_model::n_gpu_layers() const { @@ -2038,7 +2039,8 @@ llama_memory_i * llama_model::create_memory(const llama_memory_params & params, // attention KV cache for the MTP context instead of the hybrid wrapper. const bool mtp_on_hybrid_qwen35 = params.ctx_type == LLAMA_CONTEXT_TYPE_MTP && - (arch == LLM_ARCH_QWEN35 || arch == LLM_ARCH_QWEN35MOE || arch == LLM_ARCH_BAILINGMOE3); + (arch == LLM_ARCH_QWEN35 || arch == LLM_ARCH_QWEN35MOE || arch == LLM_ARCH_BAILINGMOE3 || + arch == LLM_ARCH_QWEN4EXP); const bool step35_with_mtp = arch == LLM_ARCH_STEP35 && hparams.nextn_predict_layers > 0; const bool mtp_on_step35 = diff --git a/src/models/models.h b/src/models/models.h index aea1cb1..2db61a0 100644 --- a/src/models/models.h +++ b/src/models/models.h @@ -1947,9 +1947,11 @@ struct llama_model_qwen4exp : public llama_model_base { void load_arch_hparams(llama_model_loader & ml) override; void load_arch_tensors(llama_model_loader & ml) override; - struct graph : public llm_build_delta_net_base { - graph(const llama_model & model, const llm_graph_params & params); - private: + // shared block builders: the trunk forward and the MTP draft forward run the + // same hc/attention/MoE block, so both graphs derive from this + struct graph_base : public llm_build_delta_net_base { + graph_base(const llama_model & model, const llm_graph_params & params); + // HC replaces every layer norm: residual is [n_embd, hc, n_tokens] ggml_tensor * build_hc_mix( ggml_tensor * x, @@ -1960,12 +1962,18 @@ struct llama_model_qwen4exp : public llama_model_base { ggml_tensor ** inject, int il); + // collapse the hc parallel streams [n_embd, hc, T] by their mean into [n_embd, T] + ggml_tensor * build_hc_collapse( + ggml_tensor * x, + int il); + ggml_tensor * build_hc_combine( ggml_tensor * residual, ggml_tensor * block_out, ggml_tensor * inject, int il); + // mctx_hyb null means a plain attention context with no indexer cache: dense ggml_tensor * build_layer_attn( llm_graph_input_attn_kv * inp_attn, const llama_memory_hybrid_idx_context * mctx_hyb, @@ -1994,12 +2002,18 @@ struct llama_model_qwen4exp : public llama_model_base { int * sections, int il); - ggml_tensor * build_layer_attn_linear( - llm_graph_input_rs * inp, + ggml_tensor * build_layer_ffn( ggml_tensor * cur, int il); - ggml_tensor * build_layer_ffn( + const llama_model & model; + }; + + struct graph : public graph_base { + graph(const llama_model & model, const llm_graph_params & params); + private: + ggml_tensor * build_layer_attn_linear( + llm_graph_input_rs * inp, ggml_tensor * cur, int il); @@ -2032,8 +2046,12 @@ struct llama_model_qwen4exp : public llama_model_base { std::pair build_qkvz( ggml_tensor * input, int il); + }; - const llama_model & model; + // LLM_GRAPH_TYPE_DECODER_MTP draft head: the trailing nextn block reading the + // target's wide pre-norm hidden rows + struct graph_mtp : public graph_base { + graph_mtp(const llama_model & model, const llm_graph_params & params); }; std::unique_ptr build_arch_graph(const llm_graph_params & params) const override; diff --git a/src/models/qwen4exp.cpp b/src/models/qwen4exp.cpp index ca3064d..2dd6d4f 100644 --- a/src/models/qwen4exp.cpp +++ b/src/models/qwen4exp.cpp @@ -9,6 +9,9 @@ void llama_model_qwen4exp::load_arch_hparams(llama_model_loader & ml) { ml.get_key_or_arr(LLM_KV_ROPE_DIMENSION_SECTIONS, hparams.rope_sections, 4, true); + // trailing nextn (MTP) blocks; absent in trunk-only files + ml.get_key(LLM_KV_NEXTN_PREDICT_LAYERS, hparams.nextn_predict_layers, false); + ml.get_key(LLM_KV_SSM_CONV_KERNEL, hparams.ssm_d_conv); ml.get_key(LLM_KV_SSM_INNER_SIZE, hparams.ssm_d_inner); @@ -114,7 +117,13 @@ void llama_model_qwen4exp::load_arch_tensors(llama_model_loader & ml) { { hparams.ple_head_dim, ple_rows }, 0); } - for (int il = 0; il < n_layer; ++il) { + // MTP sidecars carry only the trailing nextn block: relax the trunk tensors then + const uint32_t n_main = n_layer - hparams.nextn_predict_layers; + const bool mtp_only = (hparams.nextn_predict_layers > 0) && + (ml.get_weight("blk.0.hc_attn_norm.weight") == nullptr); + const int trunk_flags = mtp_only ? TENSOR_NOT_REQUIRED : 0; + + auto load_block_trunk = [&](int il, int flags) { auto & layer = layers[il]; const int64_t n_ff_exp = hparams.n_ff_exp ? hparams.n_ff_exp : n_ff / n_expert_used; @@ -129,50 +138,94 @@ void llama_model_qwen4exp::load_arch_tensors(llama_model_loader & ml) { const int64_t conv_dim = key_dim * 2 + value_dim; // two HC modules per layer: before the token mixer, before the MoE - layer.hc_attn_norm = create_tensor(tn(LLM_TENSOR_HC_ATTN_NORM, "weight", il), { hc_dim }, 0); - layer.hc_attn_down = create_tensor(tn(LLM_TENSOR_HC_ATTN_DOWN, "weight", il), { hc_dim, hc_lr }, 0); - layer.hc_attn_up = create_tensor(tn(LLM_TENSOR_HC_ATTN_UP, "weight", il), { hc_lr, hc_dim }, 0); - layer.hc_attn_inject = create_tensor(tn(LLM_TENSOR_HC_ATTN_INJECT, "weight", il), { hc_dim, hc }, 0); - layer.hc_ffn_norm = create_tensor(tn(LLM_TENSOR_HC_FFN_NORM, "weight", il), { hc_dim }, 0); - layer.hc_ffn_down = create_tensor(tn(LLM_TENSOR_HC_FFN_DOWN, "weight", il), { hc_dim, hc_lr }, 0); - layer.hc_ffn_up = create_tensor(tn(LLM_TENSOR_HC_FFN_UP, "weight", il), { hc_lr, hc_dim }, 0); - layer.hc_ffn_inject = create_tensor(tn(LLM_TENSOR_HC_FFN_INJECT, "weight", il), { hc_dim, hc }, 0); + layer.hc_attn_norm = create_tensor(tn(LLM_TENSOR_HC_ATTN_NORM, "weight", il), { hc_dim }, flags); + layer.hc_attn_down = create_tensor(tn(LLM_TENSOR_HC_ATTN_DOWN, "weight", il), { hc_dim, hc_lr }, flags); + layer.hc_attn_up = create_tensor(tn(LLM_TENSOR_HC_ATTN_UP, "weight", il), { hc_lr, hc_dim }, flags); + layer.hc_attn_inject = create_tensor(tn(LLM_TENSOR_HC_ATTN_INJECT, "weight", il), { hc_dim, hc }, flags); + layer.hc_ffn_norm = create_tensor(tn(LLM_TENSOR_HC_FFN_NORM, "weight", il), { hc_dim }, flags); + layer.hc_ffn_down = create_tensor(tn(LLM_TENSOR_HC_FFN_DOWN, "weight", il), { hc_dim, hc_lr }, flags); + layer.hc_ffn_up = create_tensor(tn(LLM_TENSOR_HC_FFN_UP, "weight", il), { hc_lr, hc_dim }, flags); + layer.hc_ffn_inject = create_tensor(tn(LLM_TENSOR_HC_FFN_INJECT, "weight", il), { hc_dim, hc }, flags); if (!hparams.is_recurrent(il)) { // full attention: wq holds [q|gate] interleaved per head - create_tensor_qkv(layer, il, n_embd, n_embd_head_k * n_head * 2, n_embd_k_gqa, n_embd_v_gqa, 0); - layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", il), { n_embd_head_k * n_head, n_embd }, 0); + create_tensor_qkv(layer, il, n_embd, n_embd_head_k * n_head * 2, n_embd_k_gqa, n_embd_v_gqa, flags); + layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", il), { n_embd_head_k * n_head, n_embd }, flags); - layer.attn_q_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_NORM, "weight", il), { n_embd_head_k }, 0); - layer.attn_k_norm = create_tensor(tn(LLM_TENSOR_ATTN_K_NORM, "weight", il), { n_embd_head_k }, 0); + layer.attn_q_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_NORM, "weight", il), { n_embd_head_k }, flags); + layer.attn_k_norm = create_tensor(tn(LLM_TENSOR_ATTN_K_NORM, "weight", il), { n_embd_head_k }, flags); const int64_t idx_dim = hparams.indexer_head_size; - layer.index_q_proj = create_tensor(tn(LLM_TENSOR_INDEXER_Q_PROJ, "weight", il), { n_embd, hparams.indexer_n_head * idx_dim }, 0); - layer.index_k_proj = create_tensor(tn(LLM_TENSOR_INDEXER_K_PROJ, "weight", il), { n_embd, idx_dim }, 0); - layer.index_q_norm = create_tensor(tn(LLM_TENSOR_INDEXER_Q_NORM, "weight", il), { idx_dim }, 0); - layer.index_k_norm = create_tensor(tn(LLM_TENSOR_INDEXER_K_NORM, "weight", il), { idx_dim }, 0); + layer.index_q_proj = create_tensor(tn(LLM_TENSOR_INDEXER_Q_PROJ, "weight", il), { n_embd, hparams.indexer_n_head * idx_dim }, flags); + layer.index_k_proj = create_tensor(tn(LLM_TENSOR_INDEXER_K_PROJ, "weight", il), { n_embd, idx_dim }, flags); + layer.index_q_norm = create_tensor(tn(LLM_TENSOR_INDEXER_Q_NORM, "weight", il), { idx_dim }, flags); + layer.index_k_norm = create_tensor(tn(LLM_TENSOR_INDEXER_K_NORM, "weight", il), { idx_dim }, flags); } else { - layer.wqkv = create_tensor(tn(LLM_TENSOR_ATTN_QKV, "weight", il), { n_embd, key_dim * 2 + value_dim }, 0); - layer.wqkv_gate = create_tensor(tn(LLM_TENSOR_ATTN_GATE, "weight", il), { n_embd, value_dim }, 0); - layer.ssm_conv1d = create_tensor(tn(LLM_TENSOR_SSM_CONV1D, "weight", il), { hparams.ssm_d_conv, conv_dim }, 0); - layer.ssm_dt = create_tensor(tn(LLM_TENSOR_SSM_DT, "bias", il), { hparams.ssm_dt_rank }, 0); - layer.ssm_a = create_tensor(tn(LLM_TENSOR_SSM_A_NOSCAN, il), { hparams.ssm_dt_rank }, 0); - layer.ssm_beta = create_tensor(tn(LLM_TENSOR_SSM_BETA, "weight", il), { n_embd, n_v_heads }, 0); - layer.ssm_alpha = create_tensor(tn(LLM_TENSOR_SSM_ALPHA, "weight", il), { n_embd, n_v_heads }, 0); - layer.ssm_norm = create_tensor(tn(LLM_TENSOR_SSM_NORM, "weight", il), { head_v_dim }, 0); - layer.ssm_out = create_tensor(tn(LLM_TENSOR_SSM_OUT, "weight", il), { value_dim, n_embd }, 0); + layer.wqkv = create_tensor(tn(LLM_TENSOR_ATTN_QKV, "weight", il), { n_embd, key_dim * 2 + value_dim }, flags); + layer.wqkv_gate = create_tensor(tn(LLM_TENSOR_ATTN_GATE, "weight", il), { n_embd, value_dim }, flags); + layer.ssm_conv1d = create_tensor(tn(LLM_TENSOR_SSM_CONV1D, "weight", il), { hparams.ssm_d_conv, conv_dim }, flags); + layer.ssm_dt = create_tensor(tn(LLM_TENSOR_SSM_DT, "bias", il), { hparams.ssm_dt_rank }, flags); + layer.ssm_a = create_tensor(tn(LLM_TENSOR_SSM_A_NOSCAN, il), { hparams.ssm_dt_rank }, flags); + layer.ssm_beta = create_tensor(tn(LLM_TENSOR_SSM_BETA, "weight", il), { n_embd, n_v_heads }, flags); + layer.ssm_alpha = create_tensor(tn(LLM_TENSOR_SSM_ALPHA, "weight", il), { n_embd, n_v_heads }, flags); + layer.ssm_norm = create_tensor(tn(LLM_TENSOR_SSM_NORM, "weight", il), { head_v_dim }, flags); + layer.ssm_out = create_tensor(tn(LLM_TENSOR_SSM_OUT, "weight", il), { value_dim, n_embd }, flags); } if (hparams.is_ple(il)) { - layer.ple_key = create_tensor(tn(LLM_TENSOR_PLE_KEY, "weight", il), { n_embd, hc_dim }, 0); - layer.ple_value = create_tensor(tn(LLM_TENSOR_PLE_VALUE, "weight", il), { n_embd, n_embd }, 0); - layer.ple_norm_key = create_tensor(tn(LLM_TENSOR_PLE_NORM_KEY, "weight", il), { hc_dim }, 0); - layer.ple_norm_query = create_tensor(tn(LLM_TENSOR_PLE_NORM_QUERY, "weight", il), { hc_dim }, 0); - layer.ple_norm_conv = create_tensor(tn(LLM_TENSOR_PLE_NORM_CONV, "weight", il), { hc_dim }, 0); - layer.ple_conv1d = create_tensor(tn(LLM_TENSOR_PLE_CONV1D, "weight", il), { hparams.ple_conv_kernel, hc_dim }, 0); + layer.ple_key = create_tensor(tn(LLM_TENSOR_PLE_KEY, "weight", il), { n_embd, hc_dim }, flags); + layer.ple_value = create_tensor(tn(LLM_TENSOR_PLE_VALUE, "weight", il), { n_embd, n_embd }, flags); + layer.ple_norm_key = create_tensor(tn(LLM_TENSOR_PLE_NORM_KEY, "weight", il), { hc_dim }, flags); + layer.ple_norm_query = create_tensor(tn(LLM_TENSOR_PLE_NORM_QUERY, "weight", il), { hc_dim }, flags); + layer.ple_norm_conv = create_tensor(tn(LLM_TENSOR_PLE_NORM_CONV, "weight", il), { hc_dim }, flags); + layer.ple_conv1d = create_tensor(tn(LLM_TENSOR_PLE_CONV1D, "weight", il), { hparams.ple_conv_kernel, hc_dim }, flags); } + layer.ffn_gate_inp = create_tensor(tn(LLM_TENSOR_FFN_GATE_INP, "weight", il), { n_embd, n_expert }, flags); + layer.ffn_down_exps = create_tensor(tn(LLM_TENSOR_FFN_DOWN_EXPS, "weight", il), { n_ff_exp, n_embd, n_expert }, flags); + create_tensor_gate_up_exps(layer, il, n_embd, n_ff_exp, n_expert, flags); + + layer.ffn_gate_inp_shexp = create_tensor(tn(LLM_TENSOR_FFN_GATE_INP_SHEXP, "weight", il), { n_embd }, flags); + layer.ffn_gate_shexp = create_tensor(tn(LLM_TENSOR_FFN_GATE_SHEXP, "weight", il), { n_embd, n_ff_shexp }, flags); + layer.ffn_up_shexp = create_tensor(tn(LLM_TENSOR_FFN_UP_SHEXP, "weight", il), { n_embd, n_ff_shexp }, flags); + layer.ffn_down_shexp = create_tensor(tn(LLM_TENSOR_FFN_DOWN_SHEXP, "weight", il), { n_ff_shexp, n_embd }, flags); + }; + + // the trailing nextn block: a full-attention decoder layer carrying the whole + // hyper-connection + MoE kit, plus the three nextn projection/norm tensors + auto load_block_mtp = [&](int il) { + auto & layer = layers[il]; + + GGML_ASSERT(!hparams.is_recurrent(il) && "the qwen4exp MTP block must be a full-attention layer"); + GGML_ASSERT(!hparams.is_ple(il) && "the qwen4exp MTP block carries no PLE tensors"); + + const int64_t n_ff_exp = hparams.n_ff_exp ? hparams.n_ff_exp : n_ff / n_expert_used; + const int64_t n_ff_shexp = hparams.n_ff_shexp ? hparams.n_ff_shexp : n_ff; + + layer.hc_attn_norm = create_tensor(tn(LLM_TENSOR_HC_ATTN_NORM, "weight", il), { hc_dim }, 0); + layer.hc_attn_down = create_tensor(tn(LLM_TENSOR_HC_ATTN_DOWN, "weight", il), { hc_dim, hc_lr }, 0); + layer.hc_attn_up = create_tensor(tn(LLM_TENSOR_HC_ATTN_UP, "weight", il), { hc_lr, hc_dim }, 0); + layer.hc_attn_inject = create_tensor(tn(LLM_TENSOR_HC_ATTN_INJECT, "weight", il), { hc_dim, hc }, 0); + layer.hc_ffn_norm = create_tensor(tn(LLM_TENSOR_HC_FFN_NORM, "weight", il), { hc_dim }, 0); + layer.hc_ffn_down = create_tensor(tn(LLM_TENSOR_HC_FFN_DOWN, "weight", il), { hc_dim, hc_lr }, 0); + layer.hc_ffn_up = create_tensor(tn(LLM_TENSOR_HC_FFN_UP, "weight", il), { hc_lr, hc_dim }, 0); + layer.hc_ffn_inject = create_tensor(tn(LLM_TENSOR_HC_FFN_INJECT, "weight", il), { hc_dim, hc }, 0); + + create_tensor_qkv(layer, il, n_embd, n_embd_head_k * n_head * 2, n_embd_k_gqa, n_embd_v_gqa, 0); + layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", il), { n_embd_head_k * n_head, n_embd }, 0); + + layer.attn_q_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_NORM, "weight", il), { n_embd_head_k }, 0); + layer.attn_k_norm = create_tensor(tn(LLM_TENSOR_ATTN_K_NORM, "weight", il), { n_embd_head_k }, 0); + + // present in the sidecar even though the MTP block's compress ratio is 0 + // (dense attention): load them so the file carries no unused tensors + const int64_t idx_dim = hparams.indexer_head_size; + layer.index_q_proj = create_tensor(tn(LLM_TENSOR_INDEXER_Q_PROJ, "weight", il), { n_embd, hparams.indexer_n_head * idx_dim }, 0); + layer.index_k_proj = create_tensor(tn(LLM_TENSOR_INDEXER_K_PROJ, "weight", il), { n_embd, idx_dim }, 0); + layer.index_q_norm = create_tensor(tn(LLM_TENSOR_INDEXER_Q_NORM, "weight", il), { idx_dim }, 0); + layer.index_k_norm = create_tensor(tn(LLM_TENSOR_INDEXER_K_NORM, "weight", il), { idx_dim }, 0); + layer.ffn_gate_inp = create_tensor(tn(LLM_TENSOR_FFN_GATE_INP, "weight", il), { n_embd, n_expert }, 0); layer.ffn_down_exps = create_tensor(tn(LLM_TENSOR_FFN_DOWN_EXPS, "weight", il), { n_ff_exp, n_embd, n_expert }, 0); create_tensor_gate_up_exps(layer, il, n_embd, n_ff_exp, n_expert, 0); @@ -181,13 +234,39 @@ void llama_model_qwen4exp::load_arch_tensors(llama_model_loader & ml) { layer.ffn_gate_shexp = create_tensor(tn(LLM_TENSOR_FFN_GATE_SHEXP, "weight", il), { n_embd, n_ff_shexp }, 0); layer.ffn_up_shexp = create_tensor(tn(LLM_TENSOR_FFN_UP_SHEXP, "weight", il), { n_embd, n_ff_shexp }, 0); layer.ffn_down_shexp = create_tensor(tn(LLM_TENSOR_FFN_DOWN_SHEXP, "weight", il), { n_ff_shexp, n_embd }, 0); + + // nextn-specific tensors. hnorm spans the whole hc layout, not n_embd + layer.nextn.eh_proj = create_tensor(tn(LLM_TENSOR_NEXTN_EH_PROJ, "weight", il), { 2 * n_embd, n_embd }, 0); + layer.nextn.enorm = create_tensor(tn(LLM_TENSOR_NEXTN_ENORM, "weight", il), { n_embd }, 0); + layer.nextn.hnorm = create_tensor(tn(LLM_TENSOR_NEXTN_HNORM, "weight", il), { hc_dim }, 0); + + // optional shared-head tensors present in some sidecar heads (the + // 35-tensor Q4_K_M/Q6_K/ROCmFP4 heads carry shared_head_norm); load + // them so the file carries no unconsumed tensors. shared_head_norm + // spans the hc layout like hnorm, not n_embd. + layer.nextn.embed_tokens = create_tensor(tn(LLM_TENSOR_NEXTN_EMBED_TOKENS, "weight", il), { n_embd, n_vocab }, TENSOR_NOT_REQUIRED); + layer.nextn.shared_head_head = create_tensor(tn(LLM_TENSOR_NEXTN_SHARED_HEAD_HEAD, "weight", il), { n_embd, n_vocab }, TENSOR_NOT_REQUIRED); + layer.nextn.shared_head_norm = create_tensor(tn(LLM_TENSOR_NEXTN_SHARED_HEAD_NORM, "weight", il), { hc_dim }, TENSOR_NOT_REQUIRED); + }; + + for (int i = 0; i < (int) n_main; ++i) { + load_block_trunk(i, trunk_flags); + } + for (int i = (int) n_main; i < n_layer; ++i) { + load_block_mtp(i); } } std::unique_ptr llama_model_qwen4exp::build_arch_graph(const llm_graph_params & params) const { + if (params.gtype == LLM_GRAPH_TYPE_DECODER_MTP) { + return std::make_unique(*this, params); + } return std::make_unique(*this, params); } +llama_model_qwen4exp::graph_base::graph_base(const llama_model & model, const llm_graph_params & params) : + llm_build_delta_net_base(params), model(model) {} + // Hyper-connections replace every layer norm: the state between blocks is `hc` // parallel residual streams [n_embd, hc, T]; each block reads one mixed [n_embd, T] // view and writes back through per-stream injection weights. @@ -195,7 +274,7 @@ std::unique_ptr llama_model_qwen4exp::build_arch_graph(const // low-rank down/silu/up gate with a plain mean collapse. // The mix output is [n_embd, T]; `inject` receives the [hc, T] scatter weights. -ggml_tensor * llama_model_qwen4exp::graph::build_hc_mix( +ggml_tensor * llama_model_qwen4exp::graph_base::build_hc_mix( ggml_tensor * x, ggml_tensor * w_norm, ggml_tensor * w_down, @@ -222,28 +301,39 @@ ggml_tensor * llama_model_qwen4exp::graph::build_hc_mix( ggml_tensor * gated = ggml_mul(ctx0, xn, gate); gated = ggml_reshape_3d(ctx0, gated, n_embd, hc, nt); + ggml_tensor * mixed = build_hc_collapse(gated, il); + + if (inject) { + *inject = build_lora_mm(w_inject, xn); + cb(*inject, "hc_inject", il); + } + + return mixed; +} + +ggml_tensor * llama_model_qwen4exp::graph_base::build_hc_collapse( + ggml_tensor * x, + int il) { + const int64_t hc = hparams.n_hc; + const int64_t nt = x->ne[2]; + // collapse the streams by their mean - ggml_tensor * mixed = ggml_view_2d(ctx0, gated, n_embd, nt, - ggml_row_size(gated->type, n_embd) * hc, 0); + ggml_tensor * mixed = ggml_view_2d(ctx0, x, n_embd, nt, + ggml_row_size(x->type, n_embd) * hc, 0); mixed = ggml_cont(ctx0, mixed); for (int64_t c = 1; c < hc; ++c) { - ggml_tensor * s = ggml_view_2d(ctx0, gated, n_embd, nt, - ggml_row_size(gated->type, n_embd) * hc, - ggml_row_size(gated->type, n_embd) * c); + ggml_tensor * s = ggml_view_2d(ctx0, x, n_embd, nt, + ggml_row_size(x->type, n_embd) * hc, + ggml_row_size(x->type, n_embd) * c); mixed = ggml_add(ctx0, mixed, s); } mixed = ggml_scale(ctx0, mixed, 1.0f / (float) hc); cb(mixed, "hc_mixed", il); - if (inject) { - *inject = build_lora_mm(w_inject, xn); - cb(*inject, "hc_inject", il); - } - return mixed; } -ggml_tensor * llama_model_qwen4exp::graph::build_hc_combine( +ggml_tensor * llama_model_qwen4exp::graph_base::build_hc_combine( ggml_tensor * residual, ggml_tensor * block_out, ggml_tensor * inject, @@ -267,7 +357,7 @@ ggml_tensor * llama_model_qwen4exp::graph::build_hc_combine( } llama_model_qwen4exp::graph::graph(const llama_model & model, const llm_graph_params & params) : - llm_build_delta_net_base(params), model(model) { + graph_base(model, params) { const int64_t hc = hparams.n_hc; GGML_ASSERT(hparams.n_embd_head_v() == hparams.n_embd_head_k()); @@ -344,6 +434,13 @@ llama_model_qwen4exp::graph::graph(const llama_model & model, const llm_graph_pa cb(res_hc, "l_last", il); } + // the wide residual before the final mixer is what an MTP draft consumes + // (hnorm spans the whole hc layout): export it for all tokens, like deepseek4 + if (cparams.embeddings_pre_norm) { + res->t_h_pre_norm = res_hc; + cb(res->t_h_pre_norm, "h_pre_norm", -1); + } + // the final mixer is the output norm: there is no separate one ggml_tensor * cur = build_hc_mix(res_hc, model.hc_head_norm, model.hc_head_down, model.hc_head_up, @@ -363,6 +460,132 @@ llama_model_qwen4exp::graph::graph(const llama_model & model, const llm_graph_pa ggml_build_forward_expand(gf, cur); } +// LLM_GRAPH_TYPE_DECODER_MTP draft head. One full-attention nextn block: +// hnorm(h) and enorm(embed(t)) collapse to [n_embd], concat, eh_proj, +// then the same hc/attention/MoE block the trunk runs, reusing its builders. +// The draft context is a plain KV cache (mtp_on_hybrid_qwen35), so the attention +// is dense: the nextn block's compress ratio is 0 and there is no indexer cache. +llama_model_qwen4exp::graph_mtp::graph_mtp(const llama_model & model, const llm_graph_params & params) + : graph_base(model, params) { + GGML_ASSERT(hparams.nextn_predict_layers > 0 && "qwen4exp MTP requires nextn_predict_layers > 0"); + GGML_ASSERT(hparams.nextn_predict_layers == 1 && "qwen4exp MTP currently only supports a single MTP block"); + + const int64_t hc = hparams.n_hc; + const int64_t hc_dim = hc * n_embd; + + // the MTP block lives at the source file's original layer index + const int il = (int) hparams.n_layer - (int) hparams.nextn_predict_layers; + const auto & layer = model.layers[il]; + + GGML_ASSERT(layer.nextn.eh_proj && "qwen4exp MTP block missing nextn.eh_proj"); + GGML_ASSERT(layer.nextn.enorm && "qwen4exp MTP block missing nextn.enorm"); + GGML_ASSERT(layer.nextn.hnorm && "qwen4exp MTP block missing nextn.hnorm"); + GGML_ASSERT(!hparams.is_recurrent(il) && "the qwen4exp MTP block must be a full-attention layer"); + GGML_ASSERT(!hparams.is_ple(il) && "the qwen4exp MTP block carries no PLE tensors"); + + int sections[4]; + std::copy(std::begin(hparams.rope_sections), std::begin(hparams.rope_sections) + 4, sections); + + // inputs: the accepted/next token ids plus the target's wide pre-norm hidden rows + auto inp = std::make_unique(hc_dim); + + inp->tokens = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_tokens); + ggml_set_input(inp->tokens); + + inp->embd = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, hc_dim, n_tokens); + ggml_set_input(inp->embd); + + inp->h = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, hc_dim, n_tokens); + ggml_set_input(inp->h); + ggml_set_name(inp->h, "mtp_h_input"); + + // raw views on the inputs; the unique_ptr itself moves below + ggml_tensor * inp_tokens = inp->tokens; + ggml_tensor * h_input = inp->h; + + res->add_input(std::move(inp)); + + ggml_tensor * inp_pos = build_inp_pos(); + ggml_tensor * inp_out_ids = build_inp_out_ids(); + auto * inp_attn = build_attn_inp_kv(); + + // e-side: embed the token, norm at n_embd width. The token embedding is shared by + // every stream, so broadcast it to hc copies. + ggml_tensor * e_cur = ggml_get_rows(ctx0, model.tok_embd, inp_tokens); + cb(e_cur, "mtp_tok_embd", il); + e_cur = build_norm(e_cur, layer.nextn.enorm, nullptr, LLM_NORM_RMS, il); + e_cur = ggml_repeat_4d(ctx0, + ggml_reshape_3d(ctx0, e_cur, n_embd, 1, n_tokens), + n_embd, hc, n_tokens, 1); + cb(e_cur, "mtp_enorm", il); + + // h-side: group-norm each hc stream (hnorm spans the whole hc layout). The streams + // stay distinct: the combiner runs per hyper-connection stream on the wide hidden + // state, and pooling them first is what dropped the acceptance. + ggml_tensor * h_cur = ggml_reshape_3d(ctx0, h_input, n_embd, hc, n_tokens); + h_cur = ggml_rms_norm(ctx0, h_cur, hparams.f_norm_rms_eps); + h_cur = ggml_reshape_2d(ctx0, h_cur, hc_dim, n_tokens); + h_cur = ggml_mul(ctx0, h_cur, layer.nextn.hnorm); + h_cur = ggml_reshape_3d(ctx0, h_cur, n_embd, hc, n_tokens); // JAY-FIX: restore per-stream 3D before concat + cb(h_cur, "mtp_hnorm", il); + + ggml_tensor * concat = ggml_concat(ctx0, e_cur, h_cur, /*dim=*/ 0); + cb(concat, "mtp_concat", il); + + // [2*n_embd, hc, T] @ eh_proj [2*n_embd, n_embd]: one matmul per stream, i.e. + // fc_embedding @ e + fc_hidden @ h for each of the hc streams + ggml_tensor * cur = build_lora_mm(layer.nextn.eh_proj, concat); + cb(cur, "mtp_eh_proj", il); + + // the projection is already the wide residual: hc distinct streams, not hc copies + ggml_tensor * res_hc = cur; + cb(res_hc, "mtp_hc_init", il); + + // attention sublayer + ggml_tensor * inject = nullptr; + ggml_tensor * mixed = build_hc_mix(res_hc, + layer.hc_attn_norm, layer.hc_attn_down, layer.hc_attn_up, layer.hc_attn_inject, + &inject, il); + + ggml_build_forward_expand(gf, mixed); + + cur = build_layer_attn(inp_attn, nullptr, mixed, inp_pos, sections, il); + + res_hc = build_hc_combine(res_hc, cur, inject, il); + + // MoE sublayer + mixed = build_hc_mix(res_hc, + layer.hc_ffn_norm, layer.hc_ffn_down, layer.hc_ffn_up, layer.hc_ffn_inject, + &inject, il); + + cur = build_layer_ffn(mixed, il); + cb(cur, "mtp_ffn_out", il); + + res_hc = build_hc_combine(res_hc, cur, inject, il); + + // the next draft step consumes the same wide pre-mix residual the target exports + res->t_h_pre_norm = res_hc; + cb(res_hc, "h_nextn", -1); + + // the final mixer is the output norm: the sidecar carries its own hc head + cur = build_hc_mix(res_hc, + model.hc_head_norm, model.hc_head_down, model.hc_head_up, + nullptr, nullptr, -1); + + if (inp_out_ids) { + cur = ggml_get_rows(ctx0, cur, inp_out_ids); + } + + cb(cur, "result_norm", -1); + res->t_embd = cur; + + cur = build_lora_mm(model.output, cur, model.output_s); + cb(cur, "result_output", -1); + res->t_logits = cur; + + ggml_build_forward_expand(gf, cur); +} + std::pair llama_model_qwen4exp::graph::build_qkvz( ggml_tensor * input, int il) { @@ -418,7 +641,7 @@ public: const uint32_t ratio; }; -ggml_tensor * llama_model_qwen4exp::graph::build_qsa_top_k( +ggml_tensor * llama_model_qwen4exp::graph_base::build_qsa_top_k( const llama_memory_hybrid_idx_context * mctx_hyb, ggml_tensor * cur, ggml_tensor * inp_pos, @@ -541,7 +764,7 @@ ggml_tensor * llama_model_qwen4exp::graph::build_qsa_top_k( // for llm_graph_input_attn_k_dsa, and the mask construction below is a copy of that one. // It is kept here rather than factored into a shared helper so that the attention path used // by every other architecture is untouched by this arch. -ggml_tensor * llama_model_qwen4exp::graph::build_attn_qsa( +ggml_tensor * llama_model_qwen4exp::graph_base::build_attn_qsa( llm_graph_input_attn_kv * inp, ggml_tensor * q_cur, ggml_tensor * k_cur, @@ -607,7 +830,7 @@ ggml_tensor * llama_model_qwen4exp::graph::build_attn_qsa( return cur; } -ggml_tensor * llama_model_qwen4exp::graph::build_layer_attn( +ggml_tensor * llama_model_qwen4exp::graph_base::build_layer_attn( llm_graph_input_attn_kv * inp, const llama_memory_hybrid_idx_context * mctx_hyb, ggml_tensor * cur, @@ -617,8 +840,9 @@ ggml_tensor * llama_model_qwen4exp::graph::build_layer_attn( const int64_t n_embd_head = hparams.n_embd_head_v(); GGML_ASSERT(n_embd_head == hparams.n_embd_head_k()); - // indexer reads the same block input as q/k/v; no cache or no ratio means dense - const bool qsa = mctx_hyb->get_idx() != nullptr && hparams.attn_compress_ratio[il] > 0; + // indexer reads the same block input as q/k/v; no cache or no ratio means dense. + // a null mctx_hyb is a plain attention context (the MTP draft): always dense + const bool qsa = mctx_hyb != nullptr && mctx_hyb->get_idx() != nullptr && hparams.attn_compress_ratio[il] > 0; ggml_tensor * top_k = qsa ? build_qsa_top_k(mctx_hyb, cur, inp_pos, sections, il) : nullptr; @@ -834,7 +1058,7 @@ ggml_tensor * llama_model_qwen4exp::graph::build_layer_attn_linear( return cur; } -ggml_tensor * llama_model_qwen4exp::graph::build_layer_ffn(ggml_tensor * cur, const int il) { +ggml_tensor * llama_model_qwen4exp::graph_base::build_layer_ffn(ggml_tensor * cur, const int il) { // Check if this is an MoE layer GGML_ASSERT(model.layers[il].ffn_gate_inp != nullptr);