From 1dc4a579b42905007dba3b3ef42e69fed57a9056 Mon Sep 17 00:00:00 2001 From: sudoingX <200180104+sudoingX@users.noreply.github.com> Date: Fri, 18 Sep 2026 21:36:39 +0000 Subject: [PATCH] Fix: apply the Hadamard inverse to token embeddings in the qwen35 MTP graph The qwen35 MTP draft graph looks the draft token's embedding up with a raw ggml_get_rows on model.tok_embd. Models that fold a Hadamard rotation into their weights list token_embd.weight in prism.hadamard.inverse_weight_names and store its rows rotated; the main graph undoes that in llm_graph_context::build_inp_embd(), the MTP graph did not, so llama_verify_hadamard_graph rejects the draft context. Reproduction: take a Hadamard-folded qwen35 file with a nextn block, for example Ternary-Bonsai-2-27B-PTQ1_0 with the blk.64 tensors of a Qwen 3.8 27B GGUF appended and qwen35.nextn_predict_layers = 1 (tools in github.com/sudoingX/bonsai2-small-gpu), and start llama-server -m model.gguf -ngl 99 -fa on -c 32768 --spec-type draft-mtp --spec-draft-n-max 1 Before this change: llama_verify_hadamard_graph: latent lookup 'mtp_tok_embd-64' consumed by op=RMS_NORM name='norm-64' src0 hint=0 llama_init_from_model: failed to initialize the context: Hadamard-latent table 'token_embd.weight' is read without the inverse transform common_speculative_init_result: failed to create MTP context After: the draft context is created and the head drafts with the trunk's own embedding table (acceptance 0.85 to 0.94 on Python, 0.45 to 0.55 on prose, RTX 3060 12GB), which saves the 682 MiB copy of the donor embedding table that was needed to work around it. Models without prism.hadamard keys take the same path as before. --- src/models/qwen35.cpp | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/src/models/qwen35.cpp b/src/models/qwen35.cpp index c5f816e..174f224 100644 --- a/src/models/qwen35.cpp +++ b/src/models/qwen35.cpp @@ -1,5 +1,6 @@ #include "models.h" #include "llama-memory-recurrent.h" +#include "llama-impl.h" void llama_model_qwen35::load_arch_hparams(llama_model_loader & ml) { ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps); @@ -634,6 +635,20 @@ llama_model_qwen35::graph_mtp::graph_mtp(const llama_model & model, const llm_gr ggml_tensor * tok_embd_w = layer.nextn.embed_tokens ? layer.nextn.embed_tokens : model.tok_embd; tok_embd = ggml_get_rows(ctx0, tok_embd_w, inp->tokens); + + // a Hadamard-latent embedding table (prism.hadamard.inverse_weight_names) + // stores rotated rows; restore the primal basis right after the lookup, + // the same way llm_graph_context::build_inp_embd() does, so an MTP head + // that shares the trunk's token_embd reads standard-basis embeddings + if (hadamard_inverses) { + const auto it = hadamard_inverses->find(tok_embd_w); + if (it != hadamard_inverses->end()) { + tok_embd = llama_mul_mat_hadamard(ctx0, tok_embd, it->second.rot); + if (it->second.signs) { + tok_embd = ggml_mul(ctx0, tok_embd, it->second.signs); + } + } + } } else { tok_embd = inp->embd; } -- 2.43.0