ggml-org/llama.cpp · #28127

Model: add Tencent Hy 4 (hy_v4) preview architecture support

Little0o0 · merged Sep 4, 202618 files · 1124 + / 6
conversion/__init__.py1 + / 0
@@ -124,6 +124,7 @@     "HunYuanMoEV1ForCausalLM": "hunyuan",     "HunYuanVLForConditionalGeneration": "hunyuan",     "HYV3ForCausalLM": "hunyuan",+    "HYV4ForCausalLM": "hy_v4",     "IQuestCoderForCausalLM": "llama",     "InternLM2ForCausalLM": "internlm",     "InternLM3ForCausalLM": "internlm",
conversion/base.py3 + / 0
@@ -1507,6 +1507,9 @@ def get_vocab_base_pre(self, tokenizer) -> str:         if chkhsh == "bba3b3366b646dbdded5dbc42d59598b849371afc42f7beafa914afaa5b70aa6":             # ref: https://huggingface.co/tencent/Hunyuan-4B-Instruct             res = "hunyuan-dense"+        if chkhsh == "e6ddf9c6686791c12d698d34c31ab9be1fea9af5a3d9a6909783ab382198ae1c":+            # ref: https://huggingface.co/tencent/Hy4-preview+            res = "hy_v4"         if chkhsh == "a6b57017d60e6edb4d88ecc2845188e0eb333a70357e45dcc9b53964a73bbae6":             # ref: https://huggingface.co/tiiuae/Falcon-H1-0.5B-Base             res = "falcon-h1"
conversion/hy_v4.pyadded311 + / 0
@@ -0,0 +1,311 @@+from __future__ import annotations++import re+from typing import Iterable++import torch++from .base import ModelBase, gguf, logger+from .deepseek import DeepseekV2Model+++def split_kv_b_proj(weight: torch.Tensor, n_head: int, qk_nope: int, v_head_dim: int):+    """Split kv_b_proj into k_b (transposed) and v_b, matching DeepSeek MLA absorption.++    weight: [n_head*(qk_nope+v_head_dim), kv_lora_rank].+    Returns (k_b, v_b): k_b [n_head, kv_lora_rank, qk_nope], v_b [n_head, v_head_dim, kv_lora_rank].+    """+    kv_lora = weight.shape[-1]+    assert weight.shape[0] == n_head * (qk_nope + v_head_dim)+    kv_b = weight.view(n_head, qk_nope + v_head_dim, kv_lora)+    k_b, v_b = torch.split(kv_b, [qk_nope, v_head_dim], dim=1)+    k_b = k_b.transpose(1, 2).contiguous()  # [n_head, kv_lora, qk_nope]+    return k_b, v_b.contiguous()+++def split_gate_up(weight: torch.Tensor, moe_intermediate_size: int):+    """Split a fused stacked gate_up expert tensor into (gate, up).++    weight: [n_expert, 2*moe_intermediate_size, hidden] (gate first, up second).+    Returns (gate, up) each [n_expert, moe_intermediate_size, hidden].+    """+    assert weight.shape[1] == 2 * moe_intermediate_size, f"{weight.shape[1]} != 2*{moe_intermediate_size}"+    gate = weight[:, :moe_intermediate_size, :].contiguous()+    up = weight[:, moe_intermediate_size:, :].contiguous()+    return gate, up+++@ModelBase.register("HYV4ForCausalLM")+class HYV4Model(DeepseekV2Model):+    """HY_V4: DeepSeek-V3 style MLA + MoE with iHC, a gated MLA output and a learnable sink.++    Reuses DeepseekV2Model for the vocab and the MLA metadata, but overrides the tensor mapping+    because HY_V4 ships pre-stacked / fused experts plus extra iHC, gate and sink tensors. The+    rope rows are mapped straight through (no permute) - the graph rotates consecutive pairs.++    DSA is supported: indexer weights are exported for the layers marked "full" in indexer_types.+    "shared" layers reuse the top-k of the last preceding full layer at inference time, so they+    carry no indexer weights.++    MTP (num_nextn_predict_layers) is dropped, so the GGUF cannot be used for speculative+    decoding. The reference only runs the MTP layers while training or while speculating, so they+    cannot change single-token logits.+    """++    model_arch = gguf.MODEL_ARCH.HY_V4++    # tensors a "full" indexer layer must carry+    INDEXER_SUFFIXES = frozenset({+        "self_attn.indexer.wq_b.weight",+        "self_attn.indexer.wk.weight",+        "self_attn.indexer.k_norm.weight",+        "self_attn.indexer.k_norm.bias",+        "self_attn.indexer.weights_proj.weight",+    })++    @classmethod+    def filter_tensors(cls, item):+        # drop MTP here, not in modify_tensors, so the weights are never read+        if item[0].startswith("model.mtp_layers."):+            return None+        return super().filter_tensors(item)++    def _check_indexer_hparams(self):+        for key in ("index_n_heads", "index_head_dim", "index_topk"):+            if key not in self.hparams:+                raise ValueError(f"HY_V4 has DSA layers but no {key}")++    def indexer_is_full(self) -> list[bool] | None:+        """Per-layer indexer ownership, or None when the checkpoint has no DSA.++        indexer_types entries are "full" (owns an indexer) or "shared" (reuses the preceding+        full layer's top-k). Missing indexer_types with sparse layers means every sparse layer+        owns one.+        """+        hparams = self.hparams+        n_layer = hparams["num_hidden_layers"]+        indexer_types = hparams.get("indexer_types")++        # the reference drives DSA off indexer_types alone; layer_types is only a fallback for+        # checkpoints predating it (it was renamed to deepseek_sparse_attention upstream)+        if indexer_types is None:+            layer_types = hparams.get("layer_types") or []+            sparse = {"sparse_attention", "deepseek_sparse_attention"}+            if not any(t in sparse for t in layer_types):+                return None+            if len(layer_types) < n_layer:+                raise ValueError(f"HY_V4 layer_types has {len(layer_types)} entries, need {n_layer}")+            self._check_indexer_hparams()+            return [t in sparse for t in layer_types[:n_layer]]++        self._check_indexer_hparams()++        if len(indexer_types) < n_layer:+            raise ValueError(f"HY_V4 indexer_types has {len(indexer_types)} entries, need {n_layer}")+        unknown = {t for t in indexer_types[:n_layer]} - {"full", "shared"}+        if unknown:+            raise ValueError(f"HY_V4 unknown indexer_types values: {sorted(unknown)}")+        is_full = [t == "full" for t in indexer_types[:n_layer]]+        if is_full and not is_full[0]:+            raise ValueError("HY_V4 layer 0 must be indexer_types 'full' (nothing precedes it to share)")+        return is_full++    def set_gguf_parameters(self):+        hparams = self.hparams++        # HY4 has n_group == topk_group == 1 (no group routing). Drop the keys so the base does+        # not emit expert_group_count/used; llama.cpp then takes the ungrouped MoE path.+        if hparams.get("n_group") == 1 and hparams.get("topk_group") == 1:+            hparams.pop("n_group", None)+            hparams.pop("topk_group", None)++        # HY_V4 config expresses dense/sparse layers via mlp_layer_types, but DeepseekV2Model+        # needs first_k_dense_replace. Derive it as the contiguous leading "dense" block+        # (the real config.json also carries first_k_dense_replace; prefer it when present,+        # but assert the two agree so a mismatch fails loudly).+        mlp_types = hparams.get("mlp_layer_types")+        explicit = hparams.get("first_k_dense_replace")+        derived = None+        if mlp_types is not None:+            lead = 0+            for t in mlp_types:+                if t == "dense":+                    lead += 1+                else:+                    break+            if any(t == "dense" for t in mlp_types[lead:]):+                raise NotImplementedError("HY_V4 converter expects a contiguous leading dense block")+            derived = lead+        if explicit is not None and derived is not None and explicit != derived:+            raise ValueError(+                f"HY_V4 first_k_dense_replace ({explicit}) disagrees with mlp_layer_types "+                f"leading-dense count ({derived})"+            )+        if explicit is None:+            if derived is None:+                raise ValueError("HY_V4 needs first_k_dense_replace or mlp_layer_types to place dense layers")+            hparams["first_k_dense_replace"] = derived++        # reuse DeepseekV2 MLA + MoE metadata (forces num_key_value_heads=1, writes q/kv lora,+        # key/value lengths, expert counts, weights scale/norm, rope dims, etc.)+        super().set_gguf_parameters()++        # HY4 uses DeepSeek-V3 sigmoid routing with e_score_correction_bias. The config has no+        # scoring_func key, so the base does not write a gating func; set it explicitly.+        self.gguf_writer.add_expert_gating_func(gguf.ExpertGatingFuncType.SIGMOID)++        # routed-expert SwiGLU logits clamp (only routed experts; shared/dense are not clamped,+        # so swiglu_clamp_shexp is intentionally not written). 0.0 disables the clamp.+        swiglu_limit = float(hparams.get("swiglu_limit", 0.0) or 0.0)+        if swiglu_limit > 0.0:+            self.gguf_writer.add_swiglu_clamp_exp([swiglu_limit] * self.block_count)++        # iHC (independent Hyper-Connections)+        self.gguf_writer.add_hyper_connection_count(hparams["hc_mult"])+        self.gguf_writer.add_hyper_connection_epsilon(hparams["hc_eps"])+        self.gguf_writer.add_hyper_connection_magnitude(hparams["hc_magnitude"])++        # is_full is written explicitly; the graph must not infer it from tensor presence+        is_full = self.indexer_is_full()+        if is_full is not None:+            self.gguf_writer.add_indexer_head_count(hparams["index_n_heads"])+            self.gguf_writer.add_indexer_key_length(hparams["index_head_dim"])+            self.gguf_writer.add_indexer_top_k(hparams["index_topk"])+            self.gguf_writer.add_indexer_types(is_full)+            logger.info(+                "HY_V4 DSA: %d/%d layers own an indexer (top_k=%d, n_heads=%d, head_dim=%d)",+                sum(is_full), len(is_full), hparams["index_topk"],+                hparams["index_n_heads"], hparams["index_head_dim"],+            )++        if hparams.get("num_nextn_predict_layers", 0):+            logger.warning(+                "HY_V4: dropping %d MTP (nextn) layer(s) - the reference runs them only under "+                "training / speculative decoding. This GGUF cannot be used for speculative decoding.",+                hparams["num_nextn_predict_layers"],+            )++    def prepare_tensors(self):+        # validate before the base materializes tensors, so a mismatch fails early+        is_full = self.indexer_is_full()+        if is_full is not None:+            present: dict[int, set[str]] = {}+            for name in self.model_tensors:+                m = re.match(r"model\.layers\.(\d+)\.(self_attn\.indexer\..+)$", name)+                if m:+                    present.setdefault(int(m.group(1)), set()).add(m.group(2))+            for il, expect_full in enumerate(is_full):+                seen = present.get(il, set())+                if expect_full and seen != self.INDEXER_SUFFIXES:+                    raise ValueError(+                        f"HY_V4 layer {il} is indexer_types 'full' but is missing indexer tensors: "+                        f"{sorted(self.INDEXER_SUFFIXES - seen)}"+                    )+                if not expect_full and seen:+                    raise ValueError(+                        f"HY_V4 layer {il} is indexer_types 'shared' but carries indexer tensors: "+                        f"{sorted(seen)}"+                    )++        super().prepare_tensors()++    def tensor_force_quant(self, name, new_name, bid, n_dims):+        # iHC mixing matrices are 2D .weight tensors that the reference keeps in fp32+        # (_keep_in_fp32_modules_strict). 1D tensors (hc_base/scale, attn_sinks,+        # e_score_correction_bias) and the router (FFN_GATE_INP) are already forced F32 by the+        # base rules. Force the HC *_fn matrices here.+        if new_name.endswith(("hc_attn_fn.weight", "hc_ffn_fn.weight", "output_hc_fn.weight")):+            return gguf.GGMLQuantizationType.F32+        # indexer k_norm is fp32 in the reference; the base rules already cover+        # *_norm.weight and INDEXER_PROJ, but not this bias+        if self.match_model_tensor_name(new_name, gguf.MODEL_TENSOR.INDEXER_K_NORM, bid, suffix=".bias"):+            return gguf.GGMLQuantizationType.F32+        # enable_lm_head_fp32: mirror the reference fp32 LM-head matmul by keeping output F32.+        if new_name == "output.weight" and self.hparams.get("enable_lm_head_fp32", False):+            return gguf.GGMLQuantizationType.F32+        return super().tensor_force_quant(name, new_name, bid, n_dims)++    def modify_tensors(self, data_torch: torch.Tensor, name: str, bid: int | None) -> Iterable[tuple[str, torch.Tensor]]:+        hparams = self.hparams+        n_head = hparams["num_attention_heads"]+        qk_nope = hparams["qk_nope_head_dim"]+        v_head_dim = hparams["v_head_dim"]+        moe_inter = hparams["moe_intermediate_size"]++        tn = self.format_tensor_name++        # ---- global (non per-layer) ----+        if name == "model.embed_tokens.weight":+            return [(tn(gguf.MODEL_TENSOR.TOKEN_EMBD), data_torch)]+        if name == "model.norm.weight":+            return [(tn(gguf.MODEL_TENSOR.OUTPUT_NORM), data_torch)]+        if name == "lm_head.weight":+            return [(tn(gguf.MODEL_TENSOR.OUTPUT), data_torch)]+        if name == "model.hc_head.hc_head_fn":+            return [(tn(gguf.MODEL_TENSOR.HC_HEAD_FN), data_torch)]+        if name == "model.hc_head.hc_head_base":+            return [(tn(gguf.MODEL_TENSOR.HC_HEAD_BASE), data_torch)]+        if name == "model.hc_head.hc_head_scale":+            return [(tn(gguf.MODEL_TENSOR.HC_HEAD_SCALE), data_torch)]++        assert bid is not None, f"expected a per-layer tensor, got {name!r}"++        # ---- per-layer, keyed by suffix after 'model.layers.{bid}.' ----+        suffix = name.split(f"model.layers.{bid}.", 1)[-1]++        # note: q_b_proj and kv_a_proj_with_mqa are mapped straight through (no RoPE permute),+        # the graph rotates consecutive pairs so the rows need no reordering+        simple = {+            "input_layernorm.weight":          (gguf.MODEL_TENSOR.ATTN_NORM, ".weight"),+            "post_attention_layernorm.weight": (gguf.MODEL_TENSOR.FFN_NORM,  ".weight"),+            "self_attn.q_a_proj.weight":       (gguf.MODEL_TENSOR.ATTN_Q_A, ".weight"),+            "self_attn.q_a_layernorm.weight":  (gguf.MODEL_TENSOR.ATTN_Q_A_NORM, ".weight"),+            "self_attn.q_b_proj.weight":       (gguf.MODEL_TENSOR.ATTN_Q_B, ".weight"),+            "self_attn.kv_a_proj_with_mqa.weight": (gguf.MODEL_TENSOR.ATTN_KV_A_MQA, ".weight"),+            "self_attn.kv_a_layernorm.weight": (gguf.MODEL_TENSOR.ATTN_KV_A_NORM, ".weight"),+            "self_attn.o_proj.weight":         (gguf.MODEL_TENSOR.ATTN_OUT, ".weight"),+            "self_attn.linear_gate.weight":    (gguf.MODEL_TENSOR.ATTN_GATE, ".weight"),+            "self_attn.learnable_sink_param":  (gguf.MODEL_TENSOR.ATTN_SINKS, ".weight"),+            "self_attn.indexer.wq_b.weight":   (gguf.MODEL_TENSOR.INDEXER_ATTN_Q_B, ".weight"),+            "self_attn.indexer.wk.weight":     (gguf.MODEL_TENSOR.INDEXER_ATTN_K, ".weight"),+            "self_attn.indexer.k_norm.weight": (gguf.MODEL_TENSOR.INDEXER_K_NORM, ".weight"),+            "self_attn.indexer.k_norm.bias":   (gguf.MODEL_TENSOR.INDEXER_K_NORM, ".bias"),+            "self_attn.indexer.weights_proj.weight": (gguf.MODEL_TENSOR.INDEXER_PROJ, ".weight"),+            "hc_attn_layer.hc_pre.hc_fn":      (gguf.MODEL_TENSOR.HC_ATTN_FN, ".weight"),+            "hc_attn_layer.hc_pre.hc_base":    (gguf.MODEL_TENSOR.HC_ATTN_BASE, ".weight"),+            "hc_attn_layer.hc_pre.hc_scale":   (gguf.MODEL_TENSOR.HC_ATTN_SCALE, ".weight"),+            "hc_mlp_layer.hc_pre.hc_fn":       (gguf.MODEL_TENSOR.HC_FFN_FN, ".weight"),+            "hc_mlp_layer.hc_pre.hc_base":     (gguf.MODEL_TENSOR.HC_FFN_BASE, ".weight"),+            "hc_mlp_layer.hc_pre.hc_scale":    (gguf.MODEL_TENSOR.HC_FFN_SCALE, ".weight"),+            "mlp.gate.weight":                 (gguf.MODEL_TENSOR.FFN_GATE_INP, ".weight"),+            "mlp.gate.e_score_correction.bias":(gguf.MODEL_TENSOR.FFN_EXP_PROBS_B, ".bias"),+            "mlp.gate_proj.weight":            (gguf.MODEL_TENSOR.FFN_GATE, ".weight"),+            "mlp.up_proj.weight":              (gguf.MODEL_TENSOR.FFN_UP, ".weight"),+            "mlp.down_proj.weight":            (gguf.MODEL_TENSOR.FFN_DOWN, ".weight"),+            "mlp.shared_experts.gate_proj.weight": (gguf.MODEL_TENSOR.FFN_GATE_SHEXP, ".weight"),+            "mlp.shared_experts.up_proj.weight":   (gguf.MODEL_TENSOR.FFN_UP_SHEXP, ".weight"),+            "mlp.shared_experts.down_proj.weight": (gguf.MODEL_TENSOR.FFN_DOWN_SHEXP, ".weight"),+        }+        if suffix in simple:+            key, sfx = simple[suffix]+            return [(tn(key, bid, sfx), data_torch)]++        # kv_b_proj: split into k_b (transposed) and v_b+        if suffix == "self_attn.kv_b_proj.weight":+            k_b, v_b = split_kv_b_proj(data_torch, n_head, qk_nope, v_head_dim)+            return [+                (tn(gguf.MODEL_TENSOR.ATTN_K_B, bid), k_b),+                (tn(gguf.MODEL_TENSOR.ATTN_V_B, bid), v_b),+            ]++        # fused stacked experts: split gate_up into gate/up+        if suffix == "mlp.experts.gate_up_proj":+            gate, up = split_gate_up(data_torch, moe_inter)+            return [+                (tn(gguf.MODEL_TENSOR.FFN_GATE_EXP, bid), gate),+                (tn(gguf.MODEL_TENSOR.FFN_UP_EXP, bid), up),+            ]+        if suffix == "mlp.experts.down_proj":+            return [(tn(gguf.MODEL_TENSOR.FFN_DOWN_EXP, bid), data_torch)]++        raise ValueError(f"Unsupported HY_V4 tensor {name!r} (suffix {suffix!r})")
convert_hf_to_gguf_update.py1 + / 0
@@ -176,6 +176,7 @@ class TOKENIZER_TYPE(IntEnum):     {"name": "minerva-7b", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/sapienzanlp/Minerva-7B-base-v1.0", "chkhsh": "1431a23e583c97432bc230bff598d103ddb5a1f89960c8f1d1051aaa944d0b35"},     {"name": "hunyuan", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/tencent/Hunyuan-A13B-Instruct", "chkhsh": "7e57df22b1fe23a7b1e1c7f3dc4e3f96d43a4eb0836d0c6bdc3436d7b2f1c664"},     {"name": "hunyuan-dense", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/tencent/Hunyuan-4B-Instruct", "chkhsh": "bba3b3366b646dbdded5dbc42d59598b849371afc42f7beafa914afaa5b70aa6"},+    {"name": "hy_v4", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/tencent/Hy4-preview", "chkhsh": "e6ddf9c6686791c12d698d34c31ab9be1fea9af5a3d9a6909783ab382198ae1c"},     # falcon-h1 series uses 4 different tokenizers across model sizes (0.5b - 34b), hence we need to define 4 different hashes     {"name": "falcon-h1", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/tiiuae/Falcon-H1-0.5B-Base", "chkhsh": "a6b57017d60e6edb4d88ecc2845188e0eb333a70357e45dcc9b53964a73bbae6"},     {"name": "falcon-h1", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/tiiuae/Falcon-H1-1B-Base", "chkhsh": "60476e1243776c4fb1b993dbd7a5f15ac22f83c80afdf425fa5ae01c8d44ef86"},
gguf-py/gguf/constants.py50 + / 0
@@ -230,6 +230,8 @@ class HyperConnection:         COUNT                = "{arch}.hyper_connection.count"         SINKHORN_ITERATIONS  = "{arch}.hyper_connection.sinkhorn_iterations"         EPSILON              = "{arch}.hyper_connection.epsilon"+        # scale of the post gate (DeepSeek-V4 hardcodes 2.0)+        MAGNITUDE            = "{arch}.hyper_connection.magnitude"         # absent means the mix projection is full rank (DeepSeek-V4 behaviour)         LOW_RANK             = "{arch}.hyper_connection.low_rank" @@ -592,6 +594,7 @@ class MODEL_ARCH(IntEnum):     HUNYUAN_DENSE    = auto()     HUNYUAN_VL       = auto()     HY_V3            = auto()+    HY_V4            = auto()     SMOLLM3          = auto()     GPT_OSS          = auto()     LFM2             = auto()@@ -1345,6 +1348,7 @@ class MODEL_TENSOR(IntEnum):     MODEL_ARCH.HUNYUAN_DENSE:    "hunyuan-dense",     MODEL_ARCH.HUNYUAN_VL:       "hunyuan_vl",     MODEL_ARCH.HY_V3:            "hy_v3",+    MODEL_ARCH.HY_V4:            "hy_v4",     MODEL_ARCH.SMOLLM3:          "smollm3",     MODEL_ARCH.GPT_OSS:          "gpt-oss",     MODEL_ARCH.LFM2:             "lfm2",@@ -4739,6 +4743,48 @@ class MODEL_TENSOR(IntEnum):         MODEL_TENSOR.NEXTN_SHARED_HEAD_HEAD,         MODEL_TENSOR.NEXTN_SHARED_HEAD_NORM,     ],+    MODEL_ARCH.HY_V4: [+        MODEL_TENSOR.TOKEN_EMBD,+        MODEL_TENSOR.OUTPUT_NORM,+        MODEL_TENSOR.OUTPUT,+        MODEL_TENSOR.ROPE_FREQS,+        MODEL_TENSOR.HC_HEAD_FN,+        MODEL_TENSOR.HC_HEAD_BASE,+        MODEL_TENSOR.HC_HEAD_SCALE,+        MODEL_TENSOR.ATTN_NORM,+        MODEL_TENSOR.ATTN_SINKS,+        MODEL_TENSOR.ATTN_Q_A,+        MODEL_TENSOR.ATTN_Q_A_NORM,+        MODEL_TENSOR.ATTN_Q_B,+        MODEL_TENSOR.ATTN_KV_A_MQA,+        MODEL_TENSOR.ATTN_KV_A_NORM,+        MODEL_TENSOR.ATTN_K_B,+        MODEL_TENSOR.ATTN_V_B,+        MODEL_TENSOR.ATTN_OUT,+        MODEL_TENSOR.ATTN_GATE,+        MODEL_TENSOR.INDEXER_K_NORM,+        MODEL_TENSOR.INDEXER_PROJ,+        MODEL_TENSOR.INDEXER_ATTN_K,+        MODEL_TENSOR.INDEXER_ATTN_Q_B,+        MODEL_TENSOR.HC_ATTN_FN,+        MODEL_TENSOR.HC_ATTN_BASE,+        MODEL_TENSOR.HC_ATTN_SCALE,+        MODEL_TENSOR.HC_FFN_FN,+        MODEL_TENSOR.HC_FFN_BASE,+        MODEL_TENSOR.HC_FFN_SCALE,+        MODEL_TENSOR.FFN_GATE_INP,+        MODEL_TENSOR.FFN_EXP_PROBS_B,+        MODEL_TENSOR.FFN_NORM,+        MODEL_TENSOR.FFN_GATE,+        MODEL_TENSOR.FFN_DOWN,+        MODEL_TENSOR.FFN_UP,+        MODEL_TENSOR.FFN_GATE_EXP,+        MODEL_TENSOR.FFN_DOWN_EXP,+        MODEL_TENSOR.FFN_UP_EXP,+        MODEL_TENSOR.FFN_GATE_SHEXP,+        MODEL_TENSOR.FFN_DOWN_SHEXP,+        MODEL_TENSOR.FFN_UP_SHEXP,+    ],     MODEL_ARCH.SMOLLM3: [         MODEL_TENSOR.TOKEN_EMBD,         MODEL_TENSOR.OUTPUT_NORM,@@ -5438,6 +5484,10 @@ class MODEL_TENSOR(IntEnum):         MODEL_TENSOR.ROPE_FREQS,         MODEL_TENSOR.ATTN_ROT_EMBD,     ],+    MODEL_ARCH.HY_V4: [+        MODEL_TENSOR.ROPE_FREQS,+        MODEL_TENSOR.ATTN_ROT_EMBD,+    ],     MODEL_ARCH.CHATGLM: [         MODEL_TENSOR.ROPE_FREQS,     ],
gguf-py/gguf/gguf_writer.py3 + / 0
@@ -1055,6 +1055,9 @@ def add_hyper_connection_sinkhorn_iterations(self, count: int) -> None:     def add_hyper_connection_epsilon(self, value: float) -> None:         self.add_float32(Keys.HyperConnection.EPSILON.format(arch=self.arch), value) +    def add_hyper_connection_magnitude(self, value: float) -> None:+        self.add_float32(Keys.HyperConnection.MAGNITUDE.format(arch=self.arch), value)+     def add_hyper_connection_low_rank(self, value: int) -> None:         self.add_uint32(Keys.HyperConnection.LOW_RANK.format(arch=self.arch), value) 
src/llama-arch.cpp3 + / 0
@@ -121,6 +121,7 @@ static const std::map<llm_arch, const char *> LLM_ARCH_NAMES = {     { LLM_ARCH_HUNYUAN_DENSE,    "hunyuan-dense"    },     { LLM_ARCH_HUNYUAN_VL,       "hunyuan_vl"       },     { LLM_ARCH_HY_V3,            "hy_v3"            },+    { LLM_ARCH_HY_V4,            "hy_v4"            },     { LLM_ARCH_SMOLLM3,          "smollm3"          },     { LLM_ARCH_OPENAI_MOE,       "gpt-oss"          },     { LLM_ARCH_LFM2,             "lfm2"             },@@ -294,6 +295,7 @@ static const std::map<llm_kv, const char *> LLM_KV_NAMES = {     { LLM_KV_HYPER_CONNECTION_COUNT,                 "%s.hyper_connection.count"                 },     { LLM_KV_HYPER_CONNECTION_SINKHORN_ITERATIONS,   "%s.hyper_connection.sinkhorn_iterations"   },     { LLM_KV_HYPER_CONNECTION_EPSILON,               "%s.hyper_connection.epsilon"               },+    { LLM_KV_HYPER_CONNECTION_MAGNITUDE,             "%s.hyper_connection.magnitude"             },     { LLM_KV_HYPER_CONNECTION_LOW_RANK,              "%s.hyper_connection.low_rank"              },      { LLM_KV_PLE_LAYERS,                             "%s.ple.layers"                             },@@ -1130,6 +1132,7 @@ bool llm_arch_supports_sm_tensor(const llm_arch & arch) {         case LLM_ARCH_OLMOE:         case LLM_ARCH_DEEPSEEK2:         case LLM_ARCH_DEEPSEEK32:+        case LLM_ARCH_HY_V4:         case LLM_ARCH_DOTS3NOTE:         case LLM_ARCH_GLM_DSA:         case LLM_ARCH_BITNET:
src/llama-arch.h2 + / 0
@@ -126,6 +126,7 @@ enum llm_arch {     LLM_ARCH_HUNYUAN_DENSE,     LLM_ARCH_HUNYUAN_VL,     LLM_ARCH_HY_V3,+    LLM_ARCH_HY_V4,     LLM_ARCH_SMOLLM3,     LLM_ARCH_OPENAI_MOE,     LLM_ARCH_LFM2,@@ -299,6 +300,7 @@ enum llm_kv {     LLM_KV_HYPER_CONNECTION_COUNT,     LLM_KV_HYPER_CONNECTION_SINKHORN_ITERATIONS,     LLM_KV_HYPER_CONNECTION_EPSILON,+    LLM_KV_HYPER_CONNECTION_MAGNITUDE,     LLM_KV_HYPER_CONNECTION_LOW_RANK,      LLM_KV_PLE_LAYERS,
src/llama-context.cpp2 + / 1
@@ -2317,7 +2317,8 @@ uint32_t llama_context::graph_max_nodes(uint32_t n_tokens) const {         (model.arch == LLM_ARCH_DFLASH && model.hparams.dsv4_hc_mult > 0) ||         model.arch == LLM_ARCH_NANBEIGE ||         model.arch == LLM_ARCH_MINIMAX_01 ||-        model.arch == LLM_ARCH_MINIMAX_M3) {+        model.arch == LLM_ARCH_MINIMAX_M3 ||+        model.arch == LLM_ARCH_HY_V4) {         res = std::max<uint32_t>(n_tokens * 40, 32u * model.n_tensors());     } else if (model.arch == LLM_ARCH_DFLASH && model.hparams.dflash_selector_rank > 0) {         // DFlash2's convolutions and selector are shape work rather than matmuls,
src/llama-graph.cpp5 + / 2
@@ -566,7 +566,10 @@ void llm_graph_input_attn_k_dsa::set_input(const llama_ubatch * ubatch) {      mctx->get_lid()->set_input_kq_mask(self_kq_mask_lid, ubatch, cparams.causal_attn); -    mctx->get_lid()->set_input_k_rot(self_k_rot_lid);+    // left unallocated when the indexer does not use the rotation+    if (self_k_rot_lid && self_k_rot_lid->buffer) {+        mctx->get_lid()->set_input_k_rot(self_k_rot_lid);+    } }  bool llm_graph_input_attn_k_dsa::can_reuse(const llm_graph_params & params) {@@ -2170,7 +2173,7 @@ ggml_tensor * llm_graph_context::build_moe_ffn(                     const float limit = hparams.swiglu_clamp_exp[il];                     constexpr float eps = 1e-6f;                     if (limit > eps) {-                        if (arch == LLM_ARCH_DEEPSEEK4 || (arch == LLM_ARCH_DFLASH && hparams.dsv4_hc_mult > 0)) {+                        if (arch == LLM_ARCH_DEEPSEEK4 || (arch == LLM_ARCH_DFLASH && hparams.dsv4_hc_mult > 0) || arch == LLM_ARCH_HY_V4) {                             cur = ggml_swiglu_clamp(ctx0, cur, up, limit);                         } else {                             up = ggml_clamp(ctx0, up, -limit, limit);
src/llama-hparams.h3 + / 0
@@ -289,6 +289,9 @@ struct llama_hparams {     // 0 = full rank (DeepSeek-V4)     uint32_t hc_low_rank = 0; +    // scale of the hyper-connection post gate (DeepSeek-V4 hardcodes 2.0)+    float    hc_magnitude = 0.0f;+     uint32_t ple_ngram_size      = 0;     uint32_t ple_heads_per_ngram = 0;     uint32_t ple_conv_kernel     = 0;
src/llama-model-saver.cpp1 + / 0
@@ -314,6 +314,7 @@ void llama_model_saver::add_kv_from_model() {     add_kv(LLM_KV_HYPER_CONNECTION_COUNT,               hparams.dsv4_hc_mult);     add_kv(LLM_KV_HYPER_CONNECTION_SINKHORN_ITERATIONS, hparams.dsv4_hc_sinkhorn_iters);     add_kv(LLM_KV_HYPER_CONNECTION_EPSILON,             hparams.dsv4_hc_eps);+    add_kv(LLM_KV_HYPER_CONNECTION_MAGNITUDE,           hparams.hc_magnitude);     add_kv(LLM_KV_HASH_LAYER_COUNT,                     hparams.dsv4_hash_layer_count);     add_kv(LLM_KV_HYPER_CONNECTION_LOW_RANK,             hparams.hc_low_rank); 
src/llama-model.cpp48 + / 1
@@ -288,6 +288,8 @@ static llama_model * llama_model_mapping(llm_arch arch, const llama_model_params             return new llama_model_hunyuan_dense(params);         case LLM_ARCH_HY_V3:             return new llama_model_hy_v3(params);+        case LLM_ARCH_HY_V4:+            return new llama_model_hy_v4(params);         case LLM_ARCH_SMOLLM3:             return new llama_model_smollm3(params);         case LLM_ARCH_OPENAI_MOE:@@ -2056,7 +2058,8 @@ void llama_model::print_info() const {          if (arch == LLM_ARCH_DEEPSEEK2 || arch == LLM_ARCH_DEEPSEEK2OCR ||                 arch == LLM_ARCH_DEEPSEEK32 || arch == LLM_ARCH_GLM_DSA ||-                arch == LLM_ARCH_DOTS3NOTE || arch == LLM_ARCH_MISTRAL4) {+                arch == LLM_ARCH_DOTS3NOTE || arch == LLM_ARCH_MISTRAL4 ||+                arch == LLM_ARCH_HY_V4) {             LLAMA_LOG_INFO("%s: n_layer_dense_lead    = %d\n",     __func__, hparams.n_layer_dense_lead);             LLAMA_LOG_INFO("%s: n_lora_q              = %d\n",     __func__, hparams.n_lora_q);             LLAMA_LOG_INFO("%s: n_lora_kv             = %d\n",     __func__, hparams.n_lora_kv);@@ -2325,6 +2328,48 @@ llama_memory_i * llama_model::create_memory(const llama_memory_params & params,                             nullptr);                 }             } break;+        case LLM_ARCH_HY_V4:+            {+                if (hparams.indexer_top_k == 0) {+                    // full-attention checkpoint: no indexer, so no indexer key cache+                    res = new llama_kv_cache(+                            *this,+                            hparams,+                            params.type_k,+                            params.type_v,+                            !cparams.flash_attn,+                            cparams.offload_kqv,+                            cparams.kv_unified,+                            cparams.n_ctx_seq,+                            cparams.n_seq_max,+                            1,+                            hparams.n_swa,+                            hparams.swa_type,+                            nullptr,+                            nullptr,+                            nullptr,+                            nullptr);+                } else {+                    // only "full" layers own an indexer, so the shared layers need no indexer cache+                    llama_kv_cache::layer_filter_cb filter_lid = [&](uint32_t il) { return hparams.is_indexer_full(il); };++                    res = new llama_kv_cache_dsa(+                            *this,+                            params.type_k,+                            params.type_v,+                            !cparams.flash_attn,+                            cparams.offload_kqv,+                            cparams.kv_unified,+                            cparams.n_ctx_seq,+                            cparams.n_seq_max,+                            1,+                            hparams.n_swa,+                            hparams.swa_type,+                            nullptr,+                            filter_lid,+                            nullptr);+                }+            } break;         case LLM_ARCH_DOTS3NOTE:             {                 GGML_ASSERT(hparams.swa_type != LLAMA_SWA_TYPE_NONE);@@ -2884,6 +2929,8 @@ llama_rope_type llama_model_rope_type(const llama_model * model) {         case LLM_ARCH_DOTS3NOTE:         case LLM_ARCH_NANBEIGE:         case LLM_ARCH_POCKETTTS:+        // HY_V4 rotates consecutive pairs, matching the reference implementation+        case LLM_ARCH_HY_V4:             return LLAMA_ROPE_TYPE_NORM;          // the pairs of head values are offset by n_rot/2
src/llama-vocab.cpp5 + / 0
@@ -318,6 +318,7 @@ struct llm_tokenizer_bpe : llm_tokenizer {             case LLAMA_VOCAB_PRE_TYPE_DEEPSEEK3_LLM:             case LLAMA_VOCAB_PRE_TYPE_HUNYUAN_DENSE:             case LLAMA_VOCAB_PRE_TYPE_JOYAI_LLM:+            case LLAMA_VOCAB_PRE_TYPE_HY_V4:                 regex_exprs = {                     "\\p{N}{1,3}",                     "[一-龥぀-ゟ゠-ヿ]+",@@ -2350,6 +2351,10 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {                 tokenizer_pre == "hunyuan-dense") {                 pre_type = LLAMA_VOCAB_PRE_TYPE_HUNYUAN_DENSE;                 clean_spaces = false;+            } else if (+                tokenizer_pre == "hy_v4") {+                pre_type = LLAMA_VOCAB_PRE_TYPE_HY_V4;+                clean_spaces = false;             } else if (                 tokenizer_pre == "joyai-llm") {                 pre_type = LLAMA_VOCAB_PRE_TYPE_JOYAI_LLM;
src/llama-vocab.h1 + / 0
@@ -65,6 +65,7 @@ enum llama_vocab_pre_type {     LLAMA_VOCAB_PRE_TYPE_GRANITE_EMB_MULTI = 54,     LLAMA_VOCAB_PRE_TYPE_MELLUM2           = 55,     LLAMA_VOCAB_PRE_TYPE_LAGUNA            = 56,+    LLAMA_VOCAB_PRE_TYPE_HY_V4             = 57, };  struct LLM_KV;
src/models/hy-v4.cppadded601 + / 0
@@ -0,0 +1,601 @@+#include "models.h"++#include "llama-kv-cache.h"+#include "llama-kv-cache-dsa.h"++#include <cmath>++// iHC (independent Hyper-Connections) helpers. Same layout as the DeepSeek-V4 HC, but without+// the comb/sinkhorn term: hc_fn makes only 2*hc coefficients (pre + post). The streams mix+// through the pre-reduce / post-distribute round trip instead.++static size_t hy_v4_elem_offset(const ggml_tensor * t, int64_t i) {+    return ggml_row_size(t->type, i);+}++static ggml_tensor * hy_v4_view_1d(ggml_context * ctx, ggml_tensor * t, int64_t ne0, int64_t i0) {+    return ggml_view_1d(ctx, t, ne0, hy_v4_elem_offset(t, i0));+}++static ggml_tensor * hy_v4_view_2d(ggml_context * ctx, ggml_tensor * t, int64_t ne0, int64_t ne1, int64_t i0) {+    return ggml_view_2d(ctx, t, ne0, ne1, t->nb[1], hy_v4_elem_offset(t, i0));+}++void llama_model_hy_v4::load_arch_hparams(llama_model_loader & ml) {+    ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps);+    ml.get_key(LLM_KV_LEADING_DENSE_BLOCK_COUNT,   hparams.n_layer_dense_lead, false);+    ml.get_key(LLM_KV_ATTENTION_Q_LORA_RANK,       hparams.n_lora_q);+    ml.get_key(LLM_KV_ATTENTION_KV_LORA_RANK,      hparams.n_lora_kv);+    ml.get_key(LLM_KV_ATTENTION_KEY_LENGTH_MLA,    hparams.n_embd_head_k_mla_impl);+    ml.get_key(LLM_KV_ATTENTION_VALUE_LENGTH_MLA,  hparams.n_embd_head_v_mla_impl);+    ml.get_key_or_arr(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, hparams.n_ff_exp_arr, hparams.n_layer_all);+    ml.get_key(LLM_KV_EXPERT_SHARED_COUNT,         hparams.n_expert_shared);+    ml.get_key(LLM_KV_EXPERT_WEIGHTS_SCALE,        hparams.expert_weights_scale, false);+    ml.get_key(LLM_KV_EXPERT_WEIGHTS_NORM,         hparams.expert_weights_norm,  false);+    ml.get_key(LLM_KV_EXPERT_GATING_FUNC,          hparams.expert_gating_func,   false);++    // routed-expert SwiGLU logits clamp (shared/dense experts are NOT clamped, so+    // swiglu_clamp_shexp is intentionally left at its 0 default)+    ml.get_key_or_arr(LLM_KV_SWIGLU_CLAMP_EXP, hparams.swiglu_clamp_exp, hparams.n_layer_all, false);++    ml.get_key(LLM_KV_HYPER_CONNECTION_COUNT,     hparams.dsv4_hc_mult);+    ml.get_key(LLM_KV_HYPER_CONNECTION_EPSILON,   hparams.dsv4_hc_eps);+    ml.get_key(LLM_KV_HYPER_CONNECTION_MAGNITUDE, hparams.hc_magnitude);++    // DSA is absent on the all-full_attention checkpoints, so indexer_top_k stays 0 there+    ml.get_key(LLM_KV_ATTENTION_INDEXER_HEAD_COUNT, hparams.indexer_n_head,    false);+    ml.get_key(LLM_KV_ATTENTION_INDEXER_KEY_LENGTH, hparams.indexer_head_size, false);+    ml.get_key(LLM_KV_ATTENTION_INDEXER_TOP_K,      hparams.indexer_top_k,     false);++    if (hparams.indexer_top_k > 0) {+        // the reference plumbs rms_norm_eps into the indexer k_norm LayerNorm, and build_norm+        // reads f_norm_eps for LLM_NORM+        hparams.f_norm_eps = hparams.f_norm_rms_eps;++        if (hparams.indexer_n_head == 0 || hparams.indexer_head_size <= hparams.n_rot()) {+            throw std::runtime_error("hy_v4: bad indexer head count / key length");+        }++        ml.get_key_or_arr(LLM_KV_ATTENTION_INDEXER_TYPES, hparams.is_indexer_full_impl, hparams.n_layer(), false);+        if (!hparams.is_indexer_full(0)) {+            throw std::runtime_error("hy_v4: layer 0 must own an indexer, nothing precedes it to share");+        }+    }++    GGML_ASSERT(hparams.is_mla());++    type = LLM_TYPE_UNKNOWN;+}++void llama_model_hy_v4::load_arch_tensors(llama_model_loader &) {+    LLAMA_LOAD_LOCALS;++    const int64_t n_embd_head_k_mla   = hparams.n_embd_head_k_mla();+    const int64_t n_embd_head_v_mla   = hparams.n_embd_head_v_mla();+    const int64_t n_embd_head_qk_rope = hparams.n_rot();+    const int64_t n_embd_head_qk_nope = n_embd_head_k_mla - n_embd_head_qk_rope;+    GGML_ASSERT(n_embd_head_qk_nope >= 1);++    const int64_t q_lora_rank     = hparams.n_lora_q;+    const int64_t kv_lora_rank    = hparams.n_lora_kv;+    const int64_t n_ff_exp        = hparams.n_ff_exp();+    const int64_t n_expert_shared = hparams.n_expert_shared;+    const int64_t hc              = hparams.dsv4_hc_mult;++    tok_embd    = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0);+    output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), {n_embd}, 0);+    output      = create_tensor(tn(LLM_TENSOR_OUTPUT,      "weight"), {n_embd, n_vocab}, 0);++    // global iHC head (collapses hc streams before the final norm)+    hc_head_fn    = create_tensor(tn(LLM_TENSOR_HC_HEAD_FN,    "weight"), {hc * n_embd, hc}, 0);+    hc_head_base  = create_tensor(tn(LLM_TENSOR_HC_HEAD_BASE,  "weight"), {hc}, 0);+    hc_head_scale = create_tensor(tn(LLM_TENSOR_HC_HEAD_SCALE, "weight"), {1}, 0);++    for (int i = 0; i < n_layer; ++i) {+        auto & layer = layers[i];++        layer.attn_norm     = create_tensor(tn(LLM_TENSOR_ATTN_NORM,     "weight", i), {n_embd}, 0);+        layer.attn_sinks    = create_tensor(tn(LLM_TENSOR_ATTN_SINKS,    "weight", i), {n_head}, 0);++        layer.wq_a          = create_tensor(tn(LLM_TENSOR_ATTN_Q_A,      "weight", i), {n_embd, q_lora_rank}, 0);+        layer.attn_q_a_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_A_NORM, "weight", i), {q_lora_rank}, 0);+        layer.wq_b          = create_tensor(tn(LLM_TENSOR_ATTN_Q_B,      "weight", i), {q_lora_rank, n_head * n_embd_head_k_mla}, 0);+        layer.wkv_a_mqa     = create_tensor(tn(LLM_TENSOR_ATTN_KV_A_MQA, "weight", i), {n_embd, kv_lora_rank + n_embd_head_qk_rope}, 0);+        layer.attn_kv_a_norm= create_tensor(tn(LLM_TENSOR_ATTN_KV_A_NORM,"weight", i), {kv_lora_rank}, 0);+        layer.wk_b          = create_tensor(tn(LLM_TENSOR_ATTN_K_B,      "weight", i), {n_embd_head_qk_nope, kv_lora_rank, n_head}, 0);+        layer.wv_b          = create_tensor(tn(LLM_TENSOR_ATTN_V_B,      "weight", i), {kv_lora_rank, n_embd_head_v_mla, n_head}, 0);+        layer.wo            = create_tensor(tn(LLM_TENSOR_ATTN_OUT,      "weight", i), {n_head * n_embd_head_v_mla, n_embd}, 0);+        layer.wqkv_gate     = create_tensor(tn(LLM_TENSOR_ATTN_GATE,     "weight", i), {n_embd, n_head * n_embd_head_v_mla}, 0);++        // only "full" indexer layers ship weights; "shared" layers reuse their top-k+        if (hparams.indexer_top_k > 0 && hparams.is_indexer_full(i)) {+            const int64_t n_indexer_head = hparams.indexer_n_head;+            const int64_t n_embd_indexer = hparams.indexer_head_size;++            layer.indexer_attn_q_b = create_tensor(tn(LLM_TENSOR_INDEXER_ATTN_Q_B, "weight", i), {q_lora_rank, n_indexer_head * n_embd_indexer}, 0);+            layer.indexer_attn_k   = create_tensor(tn(LLM_TENSOR_INDEXER_ATTN_K,   "weight", i), {n_embd, n_embd_indexer}, 0);+            layer.indexer_k_norm   = create_tensor(tn(LLM_TENSOR_INDEXER_K_NORM,   "weight", i), {n_embd_indexer}, 0);+            layer.indexer_k_norm_b = create_tensor(tn(LLM_TENSOR_INDEXER_K_NORM,   "bias",   i), {n_embd_indexer}, 0);+            layer.indexer_proj     = create_tensor(tn(LLM_TENSOR_INDEXER_PROJ,     "weight", i), {n_embd, n_indexer_head}, 0);+        }++        layer.hc_attn_fn    = create_tensor(tn(LLM_TENSOR_HC_ATTN_FN,    "weight", i), {hc * n_embd, 2 * hc}, 0);+        layer.hc_attn_base  = create_tensor(tn(LLM_TENSOR_HC_ATTN_BASE,  "weight", i), {2 * hc}, 0);+        layer.hc_attn_scale = create_tensor(tn(LLM_TENSOR_HC_ATTN_SCALE, "weight", i), {2}, 0);+        layer.hc_ffn_fn     = create_tensor(tn(LLM_TENSOR_HC_FFN_FN,     "weight", i), {hc * n_embd, 2 * hc}, 0);+        layer.hc_ffn_base   = create_tensor(tn(LLM_TENSOR_HC_FFN_BASE,   "weight", i), {2 * hc}, 0);+        layer.hc_ffn_scale  = create_tensor(tn(LLM_TENSOR_HC_FFN_SCALE,  "weight", i), {2}, 0);++        layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), {n_embd}, 0);++        if (i < (int) hparams.n_layer_dense_lead) {+            layer.ffn_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "weight", i), {n_embd,   n_ff}, 0);+            layer.ffn_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "weight", i), {  n_ff, n_embd}, 0);+            layer.ffn_up   = create_tensor(tn(LLM_TENSOR_FFN_UP,   "weight", i), {n_embd,   n_ff}, 0);+        } else {+            layer.ffn_gate_inp    = create_tensor(tn(LLM_TENSOR_FFN_GATE_INP,    "weight", i), {n_embd, n_expert}, 0);+            layer.ffn_exp_probs_b = create_tensor(tn(LLM_TENSOR_FFN_EXP_PROBS_B, "bias",   i), {n_expert}, TENSOR_NOT_REQUIRED);++            if (n_expert == 0) {+                throw std::runtime_error("n_expert must be > 0");+            }+            if (n_expert_used == 0) {+                throw std::runtime_error("n_expert_used must be > 0");+            }++            layer.ffn_gate_exps = create_tensor(tn(LLM_TENSOR_FFN_GATE_EXPS, "weight", i), {n_embd,   n_ff_exp, n_expert}, 0);+            layer.ffn_up_exps   = create_tensor(tn(LLM_TENSOR_FFN_UP_EXPS,   "weight", i), {n_embd,   n_ff_exp, n_expert}, 0);+            layer.ffn_down_exps = create_tensor(tn(LLM_TENSOR_FFN_DOWN_EXPS, "weight", i), {n_ff_exp, n_embd,   n_expert}, 0);++            layer.ffn_gate_shexp = create_tensor(tn(LLM_TENSOR_FFN_GATE_SHEXP, "weight", i), {n_embd, n_ff_exp * n_expert_shared}, 0);+            layer.ffn_down_shexp = create_tensor(tn(LLM_TENSOR_FFN_DOWN_SHEXP, "weight", i), {n_ff_exp * n_expert_shared, n_embd}, 0);+            layer.ffn_up_shexp   = create_tensor(tn(LLM_TENSOR_FFN_UP_SHEXP,   "weight", i), {n_embd, n_ff_exp * n_expert_shared}, 0);+        }+    }+}++std::unique_ptr<llm_graph_context> llama_model_hy_v4::build_arch_graph(const llm_graph_params & params) const {+    return std::make_unique<graph>(*this, params);+}++// reduce hc streams x[:,i,:] weighted by w[i,:] -> [n_embd, n_tokens]+// reference runs this in fp32 (inside the float() / autocast(fp32) context)+static ggml_tensor * hy_v4_hc_reduce(ggml_context * ctx0, ggml_tensor * x, ggml_tensor * w, int64_t hc, int64_t n_embd, int64_t nt, ggml_type out_type) {+    ggml_tensor * x_f32 = ggml_cast(ctx0, x, GGML_TYPE_F32);+    ggml_tensor * result = nullptr;+    for (int64_t ih = 0; ih < hc; ++ih) {+        ggml_tensor * xh = ggml_view_2d(ctx0, x_f32, n_embd, nt, x_f32->nb[2], ih * x_f32->nb[1]);+        ggml_tensor * wh = ggml_view_2d(ctx0, w, 1, nt, w->nb[1], ih * w->nb[0]);+        ggml_tensor * cur = ggml_mul(ctx0, xh, wh);+        result = result ? ggml_add(ctx0, result, cur) : cur;+    }+    return ggml_cast(ctx0, result, out_type);+}++ggml_tensor * llama_model_hy_v4::graph::build_hc_pre(+        ggml_tensor * x,+        ggml_tensor * hc_fn,+        ggml_tensor * hc_scale,+        ggml_tensor * hc_base,+        ggml_tensor ** post,+        int il) const {+    const int64_t hc  = hparams.dsv4_hc_mult;+    const int64_t nt  = x->ne[2];+    GGML_ASSERT(x->ne[0] == n_embd && x->ne[1] == hc);++    ggml_tensor * flat = ggml_reshape_2d(ctx0, x, hc * n_embd, nt);+    ggml_tensor * flat_norm = ggml_rms_norm(ctx0, flat, hparams.f_norm_rms_eps);+    ggml_tensor * mixes = ggml_mul_mat(ctx0, hc_fn, flat_norm); // [2*hc, nt]+    cb(mixes, "hc_mixes", il);++    ggml_tensor * scale_pre  = hy_v4_view_1d(ctx0, hc_scale, 1, 0);+    ggml_tensor * scale_post = hy_v4_view_1d(ctx0, hc_scale, 1, 1);+    ggml_tensor * base_pre   = hy_v4_view_1d(ctx0, hc_base, hc, 0);+    ggml_tensor * base_post  = hy_v4_view_1d(ctx0, hc_base, hc, hc);++    // pre = sigmoid(mixes[:hc]*scale_pre + base_pre) + eps+    ggml_tensor * pre = hy_v4_view_2d(ctx0, mixes, hc, nt, 0);+    pre = ggml_mul(ctx0, pre, scale_pre);+    pre = ggml_add(ctx0, pre, base_pre);+    pre = ggml_sigmoid(ctx0, pre);+    pre = ggml_scale_bias(ctx0, pre, 1.0f, hparams.dsv4_hc_eps);+    cb(pre, "hc_pre", il);++    // post = magnitude*sigmoid(mixes[hc:2hc]*scale_post + base_post) + eps+    ggml_tensor * po = hy_v4_view_2d(ctx0, mixes, hc, nt, hc);+    po = ggml_mul(ctx0, po, scale_post);+    po = ggml_add(ctx0, po, base_post);+    po = ggml_sigmoid(ctx0, po);+    po = ggml_scale(ctx0, po, hparams.hc_magnitude);+    po = ggml_scale_bias(ctx0, po, 1.0f, hparams.dsv4_hc_eps);+    *post = po;+    cb(po, "hc_post_gate", il);++    return hy_v4_hc_reduce(ctx0, x, pre, hc, n_embd, nt, x->type);+}++ggml_tensor * llama_model_hy_v4::graph::build_hc_post(+        ggml_tensor * x,+        ggml_tensor * residual,+        ggml_tensor * post,+        int il) const {+    GGML_UNUSED(il);+    const int64_t hc = hparams.dsv4_hc_mult;+    const int64_t nt = x->ne[1];+    GGML_ASSERT(x->ne[0] == n_embd);+    GGML_ASSERT(residual->ne[1] == hc);++    // reference HC post runs entirely in fp32 to avoid bf16 rounding accumulation+    // across 78 layers: post.float() * x.float() + residual.float() -> .to(dtype)+    ggml_tensor * x_f32    = ggml_cast(ctx0, x, GGML_TYPE_F32);+    ggml_tensor * post_f32 = ggml_cast(ctx0, post, GGML_TYPE_F32);+    ggml_tensor * res_f32  = ggml_cast(ctx0, residual, GGML_TYPE_F32);++    ggml_tensor * out = nullptr;+    for (int64_t i = 0; i < hc; ++i) {+        ggml_tensor * res_i  = ggml_view_2d(ctx0, res_f32, n_embd, nt, res_f32->nb[2], i * res_f32->nb[1]);+        ggml_tensor * post_i = ggml_view_2d(ctx0, post_f32, 1, nt, post_f32->nb[1], i * post_f32->nb[0]);+        ggml_tensor * cur = ggml_add(ctx0, res_i, ggml_mul(ctx0, x_f32, post_i));+        cur = ggml_reshape_3d(ctx0, cur, n_embd, 1, nt);+        out = out ? ggml_concat(ctx0, out, cur, 1) : cur;+    }++    // cast back to the original type (bf16)+    out = ggml_cast(ctx0, out, residual->type);+    return out; // [n_embd, hc, nt]+}++ggml_tensor * llama_model_hy_v4::graph::build_hc_head(+        ggml_tensor * x,+        ggml_tensor * hc_fn,+        ggml_tensor * hc_scale,+        ggml_tensor * hc_base) const {+    const int64_t hc = hparams.dsv4_hc_mult;+    const int64_t nt = x->ne[2];++    ggml_tensor * flat = ggml_reshape_2d(ctx0, x, hc * n_embd, nt);+    ggml_tensor * flat_norm = ggml_rms_norm(ctx0, flat, hparams.f_norm_rms_eps);+    ggml_tensor * mixes = ggml_mul_mat(ctx0, hc_fn, flat_norm); // [hc, nt]+    cb(mixes, "hc_head_mixes", -1);++    ggml_tensor * pre = ggml_mul(ctx0, mixes, hc_scale);+    pre = ggml_add(ctx0, pre, hc_base);+    pre = ggml_sigmoid(ctx0, pre);+    pre = ggml_scale_bias(ctx0, pre, 1.0f, hparams.dsv4_hc_eps);+    cb(pre, "hc_head_pre", -1);++    return hy_v4_hc_reduce(ctx0, x, pre, hc, n_embd, nt, x->type);+}++ggml_tensor * llama_model_hy_v4::graph::build_attention(+        const llama_model & model,+        llm_graph_input_attn_k * inp_attn,+        ggml_tensor * cur,+        ggml_tensor * inp_pos,+        float kq_scale,+        int il) const {+    const auto & layer = model.layers[il];++    const int64_t n_embd_head_k       = hparams.n_embd_head_k_mla();+    const int64_t n_embd_head_qk_rope = hparams.n_rot();+    const int64_t n_embd_head_qk_nope = n_embd_head_k - n_embd_head_qk_rope;+    const uint32_t kv_lora_rank       = hparams.n_lora_kv;++    ggml_tensor * q = ggml_mul_mat(ctx0, layer.wq_a, cur);+    q = build_norm(q, layer.attn_q_a_norm, nullptr, LLM_NORM_RMS, il);+    q = ggml_mul_mat(ctx0, layer.wq_b, q);++    ggml_tensor * q_nope = ggml_view_3d(ctx0, q, n_embd_head_qk_nope, n_head, n_tokens,+            ggml_row_size(q->type, n_embd_head_k), ggml_row_size(q->type, n_embd_head_k) * n_head, 0);+    ggml_tensor * q_pe = ggml_view_3d(ctx0, q, n_embd_head_qk_rope, n_head, n_tokens,+            ggml_row_size(q->type, n_embd_head_k), ggml_row_size(q->type, n_embd_head_k) * n_head,+            ggml_row_size(q->type, n_embd_head_qk_nope));++    ggml_tensor * kv_cmpr_pe = ggml_mul_mat(ctx0, layer.wkv_a_mqa, cur);+    ggml_tensor * kv_cmpr = ggml_view_2d(ctx0, kv_cmpr_pe, kv_lora_rank, n_tokens,+            ggml_row_size(kv_cmpr_pe->type, kv_lora_rank + n_embd_head_qk_rope), 0);+    ggml_tensor * k_pe = ggml_view_3d(ctx0, kv_cmpr_pe, n_embd_head_qk_rope, 1, n_tokens,+            ggml_row_size(kv_cmpr_pe->type, kv_lora_rank + n_embd_head_qk_rope),+            ggml_row_size(kv_cmpr_pe->type, kv_lora_rank + n_embd_head_qk_rope),+            ggml_row_size(kv_cmpr_pe->type, kv_lora_rank));++    q_pe = ggml_rope_ext(ctx0, q_pe, inp_pos, nullptr, n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,+            ext_factor, attn_factor, beta_fast, beta_slow);+    cb(q_pe, "q_pe", il);+    k_pe = ggml_rope_ext(ctx0, k_pe, inp_pos, nullptr, n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,+            ext_factor, attn_factor, beta_fast, beta_slow);+    cb(k_pe, "k_pe", il);++    kv_cmpr = build_norm(kv_cmpr, layer.attn_kv_a_norm, nullptr, LLM_NORM_RMS, il);+    cb(kv_cmpr, "kv_cmpr", il);++    // MLA absorption: q_nope @ wk_b -> compressed space+    q_nope = ggml_permute(ctx0, q_nope, 0, 2, 1, 3);+    ggml_tensor * q_nope_absorbed = ggml_mul_mat(ctx0, layer.wk_b, q_nope);+    q_nope_absorbed = ggml_permute(ctx0, q_nope_absorbed, 0, 2, 1, 3);++    // note: rope must go first for in-place context shifting in build_rope_shift()+    ggml_tensor * Qcur = ggml_concat(ctx0, q_nope_absorbed, q_pe, 0);++    kv_cmpr = ggml_reshape_3d(ctx0, kv_cmpr, kv_lora_rank, 1, n_tokens);+    ggml_tensor * Kcur = ggml_concat(ctx0, kv_cmpr, k_pe, 0);+    ggml_tensor * Vcur = kv_cmpr;++    // MLA-as-MQA; wo applied manually below so the gated-MLA gate can sit before o_proj+    ggml_tensor * attn = build_attn(inp_attn,+            nullptr, nullptr, nullptr,+            Qcur, Kcur, Vcur, nullptr, layer.attn_sinks, layer.wv_b, kq_scale, il);+    cb(attn, "attn_kqv", il); // [n_head * n_embd_head_v, n_tokens]++    // gated MLA: elementwise sigmoid gate on the decompressed attention output+    ggml_tensor * gate = ggml_mul_mat(ctx0, layer.wqkv_gate, cur);+    gate = ggml_sigmoid(ctx0, gate);+    attn = ggml_mul(ctx0, attn, gate);+    cb(attn, "attn_gated", il);++    ggml_tensor * out = build_lora_mm(layer.wo, attn);+    cb(out, "attn_out", il);++    return out;+}++ggml_tensor * llama_model_hy_v4::graph::build_indexer_top_k(+        const llama_model & model,+        llm_graph_input_attn_k_dsa * inp_attn_dsa,+        ggml_tensor * cur,+        ggml_tensor * qr,+        ggml_tensor * inp_pos,+        int il) const {+    const auto & layer = model.layers[il];++    const int64_t n_indexer_head       = hparams.indexer_n_head;+    const int64_t n_embd_indexer       = hparams.indexer_head_size;+    const int64_t n_embd_indexer_rope  = hparams.n_rot();+    const int64_t n_embd_indexer_nope  = n_embd_indexer - n_embd_indexer_rope;++    // nope rows come first, so rope only the last n_embd_indexer_rope rows, same as the MLA path+    ggml_tensor * iq = ggml_mul_mat(ctx0, layer.indexer_attn_q_b, qr);++    iq = ggml_reshape_3d(ctx0, iq, n_embd_indexer, n_indexer_head, n_tokens);++    iq = ggml_rope_ext(ctx0, iq, inp_pos, nullptr, n_rot, rope_type, n_ctx_orig, freq_base,+         freq_scale, ext_factor, attn_factor, beta_fast, beta_slow);+    iq = ggml_rope_set_offset(iq, n_embd_indexer_nope);+    cb(iq, "indexer_q", il);++    ggml_tensor * ik = ggml_mul_mat(ctx0, layer.indexer_attn_k, cur);++    ik = build_norm(ik, layer.indexer_k_norm, layer.indexer_k_norm_b, LLM_NORM, il);++    ik = ggml_reshape_3d(ctx0, ik, n_embd_indexer, 1, n_tokens);++    ik = ggml_rope_ext(ctx0, ik, inp_pos, nullptr, n_rot, rope_type, n_ctx_orig, freq_base,+         freq_scale, ext_factor, attn_factor, beta_fast, beta_slow);+    ik = ggml_rope_set_offset(ik, n_embd_indexer_nope);+    cb(ik, "indexer_k", il);++    // the reference applies a Hadamard rotation here, but it only helps its FP8 kernels.+    // it is orthogonal, so it does not change q.k and we can skip it.++    const auto * mctx_lid   = inp_attn_dsa->mctx->get_lid();+    const auto & k_idxs_lid = inp_attn_dsa->get_k_idxs_lid();+    ggml_build_forward_expand(gf, mctx_lid->cpy_k(ctx0, ik, k_idxs_lid, il));++    ggml_tensor * iw = ggml_mul_mat(ctx0, layer.indexer_proj, cur);++    ik = mctx_lid->get_k(ctx0, il);++    const auto n_stream = ik->ne[3];+    iq = ggml_view_4d(ctx0, iq, iq->ne[0], iq->ne[1], iq->ne[2]/n_stream, n_stream,+            iq->nb[1], iq->nb[2], iq->nb[3]/n_stream, 0);+    iw = ggml_view_4d(ctx0, iw, iw->ne[0], iw->ne[1]/n_stream, iw->ne[2], n_stream,+            iw->nb[1], iw->nb[2]/n_stream, iw->nb[3]/n_stream, 0);++    // fold both reference scale factors into the weights before the big score tensor+    iw = ggml_scale(ctx0, iw, 1.0f / sqrtf(float(n_embd_indexer * n_indexer_head)));++    ggml_tensor * score = nullptr;+    if (cparams.fused_lid) {+        score = ggml_lightning_indexer(ctx0, iq, ik, iw, inp_attn_dsa->get_kq_mask_lid());+        cb(score, "indexer_score", il);+        res->add_fused_node({LLM_FUSED_OP_LIGHTNING_INDEXER, score, il});+    } else {+        iq = ggml_permute(ctx0, iq, 0, 2, 1, 3);+        ik = ggml_permute(ctx0, ik, 0, 2, 1, 3);++        score = ggml_mul_mat(ctx0, ik, iq);+        score = ggml_cont(ctx0, ggml_permute(ctx0, score, 2, 1, 0, 3));+        score = ggml_relu(ctx0, score);+        score = ggml_mul(ctx0, score, iw);+        score = ggml_sum_rows(ctx0, score);+        score = ggml_cont(ctx0, ggml_permute(ctx0, score, 2, 1, 0, 3));+        score = ggml_add(ctx0, score, inp_attn_dsa->get_kq_mask_lid());+        cb(score, "indexer_score", il);+    }++    const uint32_t n_top_k = score->ne[0] < (int64_t) hparams.indexer_top_k ? score->ne[0] : hparams.indexer_top_k;++    return ggml_cont(ctx0, ggml_top_k(ctx0, score, n_top_k));+}++ggml_tensor * llama_model_hy_v4::graph::build_attention_dsa(+        const llama_model & model,+        llm_graph_input_attn_k_dsa * inp_attn_dsa,+        ggml_tensor * cur,+        ggml_tensor * inp_pos,+        ggml_tensor ** last_top_k,+        float kq_scale,+        int il) const {+    const auto & layer = model.layers[il];++    const int64_t n_embd_head_k       = hparams.n_embd_head_k_mla();+    const int64_t n_embd_head_qk_rope = hparams.n_rot();+    const int64_t n_embd_head_qk_nope = n_embd_head_k - n_embd_head_qk_rope;+    const uint32_t kv_lora_rank       = hparams.n_lora_kv;++    ggml_tensor * qr = ggml_mul_mat(ctx0, layer.wq_a, cur);+    qr = build_norm(qr, layer.attn_q_a_norm, nullptr, LLM_NORM_RMS, il);++    if (hparams.is_indexer_full(il)) {+        *last_top_k = build_indexer_top_k(model, inp_attn_dsa, cur, qr, inp_pos, il);+        cb(*last_top_k, "top_k", il);+    }+    GGML_ASSERT(*last_top_k != nullptr);++    ggml_tensor * q = ggml_mul_mat(ctx0, layer.wq_b, qr);++    ggml_tensor * q_nope = ggml_view_3d(ctx0, q, n_embd_head_qk_nope, n_head, n_tokens,+            ggml_row_size(q->type, n_embd_head_k), ggml_row_size(q->type, n_embd_head_k) * n_head, 0);+    ggml_tensor * q_pe = ggml_view_3d(ctx0, q, n_embd_head_qk_rope, n_head, n_tokens,+            ggml_row_size(q->type, n_embd_head_k), ggml_row_size(q->type, n_embd_head_k) * n_head,+            ggml_row_size(q->type, n_embd_head_qk_nope));++    ggml_tensor * kv_cmpr_pe = ggml_mul_mat(ctx0, layer.wkv_a_mqa, cur);+    ggml_tensor * kv_cmpr = ggml_view_2d(ctx0, kv_cmpr_pe, kv_lora_rank, n_tokens,+            ggml_row_size(kv_cmpr_pe->type, kv_lora_rank + n_embd_head_qk_rope), 0);+    ggml_tensor * k_pe = ggml_view_3d(ctx0, kv_cmpr_pe, n_embd_head_qk_rope, 1, n_tokens,+            ggml_row_size(kv_cmpr_pe->type, kv_lora_rank + n_embd_head_qk_rope),+            ggml_row_size(kv_cmpr_pe->type, kv_lora_rank + n_embd_head_qk_rope),+            ggml_row_size(kv_cmpr_pe->type, kv_lora_rank));++    q_pe = ggml_rope_ext(ctx0, q_pe, inp_pos, nullptr, n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,+            ext_factor, attn_factor, beta_fast, beta_slow);+    cb(q_pe, "q_pe", il);+    k_pe = ggml_rope_ext(ctx0, k_pe, inp_pos, nullptr, n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,+            ext_factor, attn_factor, beta_fast, beta_slow);+    cb(k_pe, "k_pe", il);++    kv_cmpr = build_norm(kv_cmpr, layer.attn_kv_a_norm, nullptr, LLM_NORM_RMS, il);+    cb(kv_cmpr, "kv_cmpr", il);++    q_nope = ggml_permute(ctx0, q_nope, 0, 2, 1, 3);+    ggml_tensor * q_nope_absorbed = ggml_mul_mat(ctx0, layer.wk_b, q_nope);+    q_nope_absorbed = ggml_permute(ctx0, q_nope_absorbed, 0, 2, 1, 3);++    ggml_tensor * Qcur = ggml_concat(ctx0, q_nope_absorbed, q_pe, 0);++    kv_cmpr = ggml_reshape_3d(ctx0, kv_cmpr, kv_lora_rank, 1, n_tokens);+    ggml_tensor * Kcur = ggml_concat(ctx0, kv_cmpr, k_pe, 0);+    ggml_tensor * Vcur = kv_cmpr;++    ggml_tensor * attn = build_attn(inp_attn_dsa,+            nullptr, nullptr, nullptr,+            Qcur, Kcur, Vcur, nullptr, layer.attn_sinks, layer.wv_b, *last_top_k, kq_scale, il);+    cb(attn, "attn_kqv", il);++    ggml_tensor * gate = ggml_mul_mat(ctx0, layer.wqkv_gate, cur);+    gate = ggml_sigmoid(ctx0, gate);+    attn = ggml_mul(ctx0, attn, gate);+    cb(attn, "attn_gated", il);++    ggml_tensor * out = build_lora_mm(layer.wo, attn);+    cb(out, "attn_out", il);++    return out;+}++llama_model_hy_v4::graph::graph(const llama_model & model, const llm_graph_params & params) :+    llm_graph_context(params) {+    const int64_t hc = hparams.dsv4_hc_mult;+    const int64_t n_embd_head_k = hparams.n_embd_head_k_mla();+    const float kq_scale = 1.0f / sqrtf(float(n_embd_head_k));++    ggml_tensor * cur;++    const bool is_dsa = hparams.indexer_top_k > 0;++    ggml_tensor * inp = build_inp_embd(model.tok_embd);+    ggml_tensor * inp_pos = build_inp_pos();+    llm_graph_input_attn_k     * inp_attn     = is_dsa ? nullptr : build_attn_inp_k();+    llm_graph_input_attn_k_dsa * inp_attn_dsa = is_dsa ? build_attn_inp_k_dsa() : nullptr;+    ggml_tensor * inp_out_ids = build_inp_out_ids();++    // top-k of the last "full" indexer layer, reused by the following "shared" layers+    ggml_tensor * last_top_k = nullptr;++    // expand the single embedding into hc parallel residual streams+    ggml_tensor * inpL = ggml_reshape_3d(ctx0, inp, n_embd, 1, n_tokens);+    inpL = ggml_repeat_4d(ctx0, inpL, n_embd, hc, n_tokens, 1);+    cb(inpL, "hc_init", -1);++    for (int il = 0; il < n_layer; ++il) {+        ggml_tensor * residual = inpL;+        ggml_tensor * post = nullptr;++        cur = build_hc_pre(inpL, model.layers[il].hc_attn_fn, model.layers[il].hc_attn_scale,+                model.layers[il].hc_attn_base, &post, il);+        cur = build_norm(cur, model.layers[il].attn_norm, nullptr, LLM_NORM_RMS, il);+        cb(cur, "attn_norm", il);++        cur = is_dsa+            ? build_attention_dsa(model, inp_attn_dsa, cur, inp_pos, &last_top_k, kq_scale, il)+            : build_attention(model, inp_attn, cur, inp_pos, kq_scale, il);++        inpL = build_hc_post(cur, residual, post, il);+        cb(inpL, "hc_attn_out", il);++        residual = inpL;+        cur = build_hc_pre(inpL, model.layers[il].hc_ffn_fn, model.layers[il].hc_ffn_scale,+                model.layers[il].hc_ffn_base, &post, il);+        cur = build_norm(cur, model.layers[il].ffn_norm, nullptr, LLM_NORM_RMS, il);+        cb(cur, "ffn_norm", il);++        const auto & layer = model.layers[il];+        if ((uint32_t) il < hparams.n_layer_dense_lead) {+            cur = build_ffn(cur,+                    layer.ffn_up,   NULL, NULL,+                    layer.ffn_gate, NULL, NULL,+                    layer.ffn_down, NULL, NULL,+                    NULL, LLM_FFN_SILU, LLM_FFN_PAR, il);+            cb(cur, "ffn_out", il);+        } else {+            ggml_tensor * moe_out = build_moe_ffn(cur,+                    layer.ffn_gate_inp,+                    layer.ffn_up_exps,+                    layer.ffn_gate_exps,+                    layer.ffn_down_exps,+                    layer.ffn_exp_probs_b,+                    n_expert, n_expert_used,+                    LLM_FFN_SILU, hparams.expert_weights_norm,+                    hparams.expert_weights_scale,+                    (llama_expert_gating_func_type) hparams.expert_gating_func,+                    il,+                    nullptr,+                    nullptr);+            cb(moe_out, "ffn_moe_out", il);++            ggml_tensor * ffn_shexp = build_ffn(cur,+                    layer.ffn_up_shexp,   NULL, NULL,+                    layer.ffn_gate_shexp, NULL, NULL,+                    layer.ffn_down_shexp, NULL, NULL,+                    NULL, LLM_FFN_SILU, LLM_FFN_PAR, il);+            cb(ffn_shexp, "ffn_shexp", il);++            cur = ggml_add(ctx0, moe_out, ffn_shexp);+            cb(cur, "ffn_out", il);+        }++        inpL = build_hc_post(cur, residual, post, il);+        cb(inpL, "l_out", il);+    }++    // prune to the requested output rows once, after all HC streams are done+    if (inp_out_ids) {+        ggml_tensor * flat = ggml_reshape_2d(ctx0, inpL, n_embd * hc, n_tokens);+        flat = ggml_get_rows(ctx0, flat, inp_out_ids);+        inpL = ggml_reshape_3d(ctx0, flat, n_embd, hc, n_outputs);+    }++    cur = build_hc_head(inpL, model.hc_head_fn, model.hc_head_scale, model.hc_head_base);+    cb(cur, "hc_head", -1);++    cur = build_norm(cur, model.output_norm, nullptr, LLM_NORM_RMS, -1);+    cb(cur, "result_norm", -1);+    res->t_embd = cur;++    cur = ggml_mul_mat(ctx0, model.output, cur);+    cb(cur, "result_output", -1);+    res->t_logits = cur;++    ggml_build_forward_expand(gf, cur);+}
src/models/models.h63 + / 0
@@ -1981,6 +1981,69 @@ struct llama_model_hy_v3 : public llama_model_base { };  +struct llama_model_hy_v4 : public llama_model_base {+    llama_model_hy_v4(const struct llama_model_params & params) : llama_model_base(params) {}+    void load_arch_hparams(llama_model_loader & ml) override;+    void load_arch_tensors(llama_model_loader & ml) override;++    struct graph : public llm_graph_context {+        graph(const llama_model & model, const llm_graph_params & params);++        // iHC (independent Hyper-Connections): pre reduces the hc streams to one and returns the+        // per-stream post gates, post writes the sublayer output back into the streams, head+        // collapses the streams before the final norm.+        ggml_tensor * build_hc_pre(+                ggml_tensor * x,+                ggml_tensor * hc_fn,+                ggml_tensor * hc_scale,+                ggml_tensor * hc_base,+                ggml_tensor ** post,+                int il) const;++        ggml_tensor * build_hc_post(+                ggml_tensor * x,+                ggml_tensor * residual,+                ggml_tensor * post,+                int il) const;++        ggml_tensor * build_hc_head(+                ggml_tensor * x,+                ggml_tensor * hc_fn,+                ggml_tensor * hc_scale,+                ggml_tensor * hc_base) const;++        ggml_tensor * build_attention(+                const llama_model & model,+                llm_graph_input_attn_k * inp_attn,+                ggml_tensor * cur,+                ggml_tensor * inp_pos,+                float kq_scale,+                int il) const;++        // DSA lightning indexer: top-k KV positions for this layer. Only "full" layers compute+        // it, "shared" layers reuse the last preceding full layer result through last_top_k.+        ggml_tensor * build_indexer_top_k(+                const llama_model & model,+                llm_graph_input_attn_k_dsa * inp_attn_dsa,+                ggml_tensor * cur,+                ggml_tensor * qr,+                ggml_tensor * inp_pos,+                int il) const;++        ggml_tensor * build_attention_dsa(+                const llama_model & model,+                llm_graph_input_attn_k_dsa * inp_attn_dsa,+                ggml_tensor * cur,+                ggml_tensor * inp_pos,+                ggml_tensor ** last_top_k,+                float kq_scale,+                int il) const;+    };++    std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;+};++ struct llama_model_hunyuan_vl : public llama_model_base {     llama_model_hunyuan_vl(const struct llama_model_params & params) : llama_model_base(params) {}     void load_arch_hparams(llama_model_loader & ml) override;
tests/test-llama-archs.cpp21 + / 2
@@ -118,7 +118,8 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {             || arch == LLM_ARCH_KIMI_LINEAR             || arch == LLM_ARCH_BAILINGMOE3             || arch == LLM_ARCH_KIMI_K3-            || arch == LLM_ARCH_MISTRAL4) {+            || arch == LLM_ARCH_MISTRAL4+            || arch == LLM_ARCH_HY_V4) {         n_embd = 128;         n_head = 1;         n_ff   = 192;@@ -191,7 +192,8 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {             || arch == LLM_ARCH_KIMI_LINEAR             || arch == LLM_ARCH_BAILINGMOE3             || arch == LLM_ARCH_KIMI_K3-            || arch == LLM_ARCH_MISTRAL4) {+            || arch == LLM_ARCH_MISTRAL4+            || arch == LLM_ARCH_HY_V4) {         ms.add_kv(LLM_KV_ATTENTION_KEY_LENGTH,       uint32_t(576));         ms.add_kv(LLM_KV_ATTENTION_VALUE_LENGTH,     uint32_t(512));         ms.add_kv(LLM_KV_ROPE_DIMENSION_COUNT,       uint32_t(64));@@ -291,6 +293,22 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {     ms.add_kv(LLM_KV_ATTENTION_INDEXER_LOCAL_BLOCKS, uint32_t(1));     ms.add_kv(LLM_KV_ROPE_DIMENSION_SECTIONS, std::vector<uint32_t>({n_embd_head/4, n_embd_head/4, n_embd_head/4, n_embd_head/4})); +    if (arch == LLM_ARCH_HY_V4) {+        ms.add_kv(LLM_KV_HYPER_CONNECTION_COUNT,     uint32_t(4));+        ms.add_kv(LLM_KV_HYPER_CONNECTION_EPSILON,   1.0e-6f);+        ms.add_kv(LLM_KV_HYPER_CONNECTION_MAGNITUDE, 2.0f);+        ms.add_kv(LLM_KV_SWIGLU_CLAMP_EXP,           10.0f);+        ms.add_kv(LLM_KV_EXPERT_WEIGHTS_SCALE,       1.0f);+        ms.add_kv(LLM_KV_EXPERT_WEIGHTS_NORM,        true);+        // layer 0 must own an indexer, the odd layers share it+        std::vector<uint32_t> indexer_types;+        indexer_types.reserve(n_layer);+        for (uint32_t il = 0; il < n_layer; il++) {+            indexer_types.push_back(il % 2 ? 0 : 1);+        }+        ms.add_kv(LLM_KV_ATTENTION_INDEXER_TYPES, indexer_types);+    }+     if (arch == LLM_ARCH_DEEPSEEK4) {         ms.add_kv(LLM_KV_ATTENTION_OUTPUT_GROUP_COUNT,         uint32_t(8));         ms.add_kv(LLM_KV_ATTENTION_OUTPUT_LORA_RANK,           uint32_t(32));@@ -468,6 +486,7 @@ static bool moe_mandatory(const llm_arch arch) {         case LLM_ARCH_ERNIE4_5_MOE:         case LLM_ARCH_HUNYUAN_MOE:         case LLM_ARCH_HY_V3:+        case LLM_ARCH_HY_V4:         case LLM_ARCH_OPENAI_MOE:         case LLM_ARCH_LFM2MOE:         case LLM_ARCH_SMALLTHINKER: