if (!layer.ssm_beta_s && layer.ssm_beta) {
layer.ssm_beta_s = create_tensor(tn(LLM_TENSOR_SSM_BETA, "scale", i), {1}, TENSOR_NOT_REQUIRED);
}
+ if (!layer.nextn.eh_proj_s && layer.nextn.eh_proj) {
+ layer.nextn.eh_proj_s = create_tensor(tn(LLM_TENSOR_NEXTN_EH_PROJ, "scale", i), {1}, TENSOR_NOT_REQUIRED);
+ }
+ if (!layer.nextn.shared_head_head_s && layer.nextn.shared_head_head) {
+ layer.nextn.shared_head_head_s = create_tensor(tn(LLM_TENSOR_NEXTN_SHARED_HEAD_HEAD, "scale", i), {1}, TENSOR_NOT_REQUIRED);
+ }
// input scales
if (!layer.wq_in_s && layer.wq) {
if (!layer.ssm_beta_in_s && layer.ssm_beta) {
layer.ssm_beta_in_s = create_tensor(tn(LLM_TENSOR_SSM_BETA, "input_scale", i), {1}, TENSOR_NOT_REQUIRED);
}
+ if (!layer.nextn.eh_proj_in_s && layer.nextn.eh_proj) {
+ layer.nextn.eh_proj_in_s = create_tensor(tn(LLM_TENSOR_NEXTN_EH_PROJ, "input_scale", i), {1}, TENSOR_NOT_REQUIRED);
+ }
+ if (!layer.nextn.shared_head_head_in_s && layer.nextn.shared_head_head) {
+ layer.nextn.shared_head_head_in_s = create_tensor(tn(LLM_TENSOR_NEXTN_SHARED_HEAD_HEAD, "input_scale", i), {1}, TENSOR_NOT_REQUIRED);
+ }
}
// output scales
if (output && output->type == GGML_TYPE_NVFP4) {
};
struct llama_layer_nextn {
- struct ggml_tensor * eh_proj = nullptr;
- struct ggml_tensor * embed_tokens = nullptr;
- struct ggml_tensor * enorm = nullptr;
- struct ggml_tensor * hnorm = nullptr;
- struct ggml_tensor * shared_head_head = nullptr;
- struct ggml_tensor * shared_head_norm = nullptr;
+ struct ggml_tensor * eh_proj = nullptr;
+ struct ggml_tensor * eh_proj_s = nullptr;
+ struct ggml_tensor * eh_proj_in_s = nullptr;
+ struct ggml_tensor * embed_tokens = nullptr;
+ struct ggml_tensor * enorm = nullptr;
+ struct ggml_tensor * hnorm = nullptr;
+ struct ggml_tensor * shared_head_head = nullptr;
+ struct ggml_tensor * shared_head_head_s = nullptr;
+ struct ggml_tensor * shared_head_head_in_s = nullptr;
+ struct ggml_tensor * shared_head_norm = nullptr;
};
struct llama_layer {
ggml_tensor * concat = ggml_concat(ctx0, e_norm, h_norm, /*dim=*/ 0);
cb(concat, "mtp_concat", il);
- ggml_tensor * cur = build_lora_mm(layer.nextn.eh_proj, concat);
+ ggml_tensor * cur = build_lora_mm(layer.nextn.eh_proj, concat, layer.nextn.eh_proj_s);
cb(cur, "mtp_eh_proj", il);
ggml_tensor * inpSA = cur;
cb(cur, "mtp_shared_head_norm", -1);
ggml_tensor * head_w = layer.nextn.shared_head_head ? layer.nextn.shared_head_head : model.output;
+ ggml_tensor * head_s = layer.nextn.shared_head_head ? layer.nextn.shared_head_head_s : model.output_s;
GGML_ASSERT(head_w && "QWEN35 MTP: missing LM head (nextn.shared_head_head or model.output)");
- cur = build_lora_mm(head_w, cur);
+ cur = build_lora_mm(head_w, cur, head_s);
cb(cur, "result_output", -1);
res->t_logits = cur;
ggml_tensor * concat = ggml_concat(ctx0, e_norm, h_norm, /*dim=*/ 0);
cb(concat, "mtp_concat", il);
- ggml_tensor * cur = build_lora_mm(layer.nextn.eh_proj, concat);
+ ggml_tensor * cur = build_lora_mm(layer.nextn.eh_proj, concat, layer.nextn.eh_proj_s);
cb(cur, "mtp_eh_proj", il);
ggml_tensor * inpSA = cur;
cb(cur, "mtp_shared_head_norm", -1);
ggml_tensor * head_w = layer.nextn.shared_head_head ? layer.nextn.shared_head_head : model.output;
+ ggml_tensor * head_s = layer.nextn.shared_head_head ? layer.nextn.shared_head_head_s : model.output_s;
GGML_ASSERT(head_w && "QWEN35MOE MTP: missing LM head (nextn.shared_head_head or model.output)");
- cur = build_lora_mm(head_w, cur);
+ cur = build_lora_mm(head_w, cur, head_s);
cb(cur, "result_output", -1);
res->t_logits = cur;