Commit a11f57ba9 for llama.cpp

commit a11f57ba93797579a5d1855ee216a31f10242676
Author: Hrishith Thadicherla <99313418+hthadicherla@users.noreply.github.com>
Date:   Thu Oct 8 21:12:28 2026 +0530

    model : fix DFlash output head sharing (#30111)

    * llama : fix DFlash output head sharing

    Assisted-by: Codex

    * dflash : read tied output weights from GGUF metadata

    Assisted-by: Codex

    * llama : share tied word embedding metadata

    Assisted-by: Codex

    * llama : remove DFlash embedding head fallback

    Assisted-by: Codex

diff --git a/conversion/gemma.py b/conversion/gemma.py
index 2a1f6931f..924d8e784 100644
--- a/conversion/gemma.py
+++ b/conversion/gemma.py
@@ -849,8 +849,8 @@ class Gemma4DSparkModel(DFlashModel):
             raise ValueError("Gemma4 DSpark attention bias and MoE are not supported")
         if (self.hparams.get("draft_vocab_size") or self.hparams["vocab_size"]) != self.hparams["vocab_size"]:
             raise ValueError("Gemma4 DSpark currently requires a full draft vocabulary")
-        if "model.lm_head.weight" not in self.model_tensors and self.hparams.get("tie_word_embeddings") is not True:
-            raise ValueError("Gemma4 DSpark requires lm_head.weight unless tie_word_embeddings is true")
+        if "model.lm_head.weight" not in self.model_tensors:
+            raise ValueError("Gemma4 DSpark requires lm_head.weight")

         self.dflash_config = self.hparams.get("dflash_config", {})
         markov_type = self.dflash_config.get("markov_head_type", self.hparams.get("markov_head_type", "vanilla"))
diff --git a/src/models/dflash.cpp b/src/models/dflash.cpp
index c8c2895b6..b6cb6bfb9 100644
--- a/src/models/dflash.cpp
+++ b/src/models/dflash.cpp
@@ -167,9 +167,6 @@ void llama_model_dflash::load_arch_tensors(llama_model_loader &) {
     // optional: reduced-vocab drafts ship their own lm head, full-vocab drafts can share the target's via ctx_other
     // a draft with its own embeddings + head references no target tensors and can run on devices the target does not use (e.g. -devd with a tensor-split target)
     output   = create_tensor(tn(LLM_TENSOR_OUTPUT,     "weight"), { n_embd, n_vocab_draft }, TENSOR_NOT_REQUIRED);
-    if (output == nullptr && tok_embd != nullptr) {
-        output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), { n_embd, n_vocab_draft }, TENSOR_DUPLICATED);
-    }

     if (hparams.dsv4_hc_mult > 0) {
         const int64_t q_lora_rank     = hparams.n_lora_q;