From d23c47f2a9175c514556fc2fc69b2e670f37ae2c Mon Sep 17 00:00:00 2001 From: fairydreaming <166155368+fairydreaming@users.noreply.github.com> Date: Mon, 7 Sep 2026 15:20:58 +0200 Subject: [PATCH] convert : refactor Hy4-preview conversion - move HC tensor mapping to the global map (#28451) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-authored-by: Stanisław Szymczyk --- conversion/hy_v4.py | 111 +++++++-------------------------- gguf-py/gguf/tensor_mapping.py | 38 +++++++++++ 2 files changed, 60 insertions(+), 89 deletions(-) diff --git a/conversion/hy_v4.py b/conversion/hy_v4.py index f564b9ec2..358e21fe5 100644 --- a/conversion/hy_v4.py +++ b/conversion/hy_v4.py @@ -9,20 +9,6 @@ from .base import ModelBase, gguf, logger from .deepseek import DeepseekV2Model -def split_kv_b_proj(weight: torch.Tensor, n_head: int, qk_nope: int, v_head_dim: int): - """Split kv_b_proj into k_b (transposed) and v_b, matching DeepSeek MLA absorption. - - weight: [n_head*(qk_nope+v_head_dim), kv_lora_rank]. - Returns (k_b, v_b): k_b [n_head, kv_lora_rank, qk_nope], v_b [n_head, v_head_dim, kv_lora_rank]. - """ - kv_lora = weight.shape[-1] - assert weight.shape[0] == n_head * (qk_nope + v_head_dim) - kv_b = weight.view(n_head, qk_nope + v_head_dim, kv_lora) - k_b, v_b = torch.split(kv_b, [qk_nope, v_head_dim], dim=1) - k_b = k_b.transpose(1, 2).contiguous() # [n_head, kv_lora, qk_nope] - return k_b, v_b.contiguous() - - def split_gate_up(weight: torch.Tensor, moe_intermediate_size: int): """Split a fused stacked gate_up expert tensor into (gate, up). @@ -36,6 +22,7 @@ def split_gate_up(weight: torch.Tensor, moe_intermediate_size: int): @ModelBase.register("HYV4ForCausalLM") +@ModelBase.example("tencent/Hy4-preview") class HYV4Model(DeepseekV2Model): """HY_V4: DeepSeek-V3 style MLA + MoE with iHC, a gated MLA output and a learnable sink. @@ -54,6 +41,8 @@ class HYV4Model(DeepseekV2Model): model_arch = gguf.MODEL_ARCH.HY_V4 + merge_expert = False + # tensors a "full" indexer layer must carry INDEXER_SUFFIXES = frozenset({ "self_attn.indexer.wq_b.weight", @@ -186,6 +175,10 @@ class HYV4Model(DeepseekV2Model): ) def prepare_tensors(self): + # Hy4-preview for some reason has num_key_value_heads equal to 8, so override it here + # without this conversion/deepseek.py fails on assert + self.hparams["num_key_value_heads"] = self.hparams["num_attention_heads"] + # validate before the base materializes tensors, so a mismatch fails early is_full = self.indexer_is_full() if is_full is not None: @@ -227,85 +220,25 @@ class HYV4Model(DeepseekV2Model): def modify_tensors(self, data_torch: torch.Tensor, name: str, bid: int | None) -> Iterable[tuple[str, torch.Tensor]]: hparams = self.hparams - n_head = hparams["num_attention_heads"] - qk_nope = hparams["qk_nope_head_dim"] - v_head_dim = hparams["v_head_dim"] moe_inter = hparams["moe_intermediate_size"] tn = self.format_tensor_name - # ---- global (non per-layer) ---- - if name == "model.embed_tokens.weight": - return [(tn(gguf.MODEL_TENSOR.TOKEN_EMBD), data_torch)] - if name == "model.norm.weight": - return [(tn(gguf.MODEL_TENSOR.OUTPUT_NORM), data_torch)] - if name == "lm_head.weight": - return [(tn(gguf.MODEL_TENSOR.OUTPUT), data_torch)] - if name == "model.hc_head.hc_head_fn": - return [(tn(gguf.MODEL_TENSOR.HC_HEAD_FN), data_torch)] - if name == "model.hc_head.hc_head_base": - return [(tn(gguf.MODEL_TENSOR.HC_HEAD_BASE), data_torch)] - if name == "model.hc_head.hc_head_scale": - return [(tn(gguf.MODEL_TENSOR.HC_HEAD_SCALE), data_torch)] - - assert bid is not None, f"expected a per-layer tensor, got {name!r}" - - # ---- per-layer, keyed by suffix after 'model.layers.{bid}.' ---- - suffix = name.split(f"model.layers.{bid}.", 1)[-1] - - # note: q_b_proj and kv_a_proj_with_mqa are mapped straight through (no RoPE permute), - # the graph rotates consecutive pairs so the rows need no reordering - simple = { - "input_layernorm.weight": (gguf.MODEL_TENSOR.ATTN_NORM, ".weight"), - "post_attention_layernorm.weight": (gguf.MODEL_TENSOR.FFN_NORM, ".weight"), - "self_attn.q_a_proj.weight": (gguf.MODEL_TENSOR.ATTN_Q_A, ".weight"), - "self_attn.q_a_layernorm.weight": (gguf.MODEL_TENSOR.ATTN_Q_A_NORM, ".weight"), - "self_attn.q_b_proj.weight": (gguf.MODEL_TENSOR.ATTN_Q_B, ".weight"), - "self_attn.kv_a_proj_with_mqa.weight": (gguf.MODEL_TENSOR.ATTN_KV_A_MQA, ".weight"), - "self_attn.kv_a_layernorm.weight": (gguf.MODEL_TENSOR.ATTN_KV_A_NORM, ".weight"), - "self_attn.o_proj.weight": (gguf.MODEL_TENSOR.ATTN_OUT, ".weight"), - "self_attn.linear_gate.weight": (gguf.MODEL_TENSOR.ATTN_GATE, ".weight"), - "self_attn.learnable_sink_param": (gguf.MODEL_TENSOR.ATTN_SINKS, ".weight"), - "self_attn.indexer.wq_b.weight": (gguf.MODEL_TENSOR.INDEXER_ATTN_Q_B, ".weight"), - "self_attn.indexer.wk.weight": (gguf.MODEL_TENSOR.INDEXER_ATTN_K, ".weight"), - "self_attn.indexer.k_norm.weight": (gguf.MODEL_TENSOR.INDEXER_K_NORM, ".weight"), - "self_attn.indexer.k_norm.bias": (gguf.MODEL_TENSOR.INDEXER_K_NORM, ".bias"), - "self_attn.indexer.weights_proj.weight": (gguf.MODEL_TENSOR.INDEXER_PROJ, ".weight"), - "hc_attn_layer.hc_pre.hc_fn": (gguf.MODEL_TENSOR.HC_ATTN_FN, ".weight"), - "hc_attn_layer.hc_pre.hc_base": (gguf.MODEL_TENSOR.HC_ATTN_BASE, ".weight"), - "hc_attn_layer.hc_pre.hc_scale": (gguf.MODEL_TENSOR.HC_ATTN_SCALE, ".weight"), - "hc_mlp_layer.hc_pre.hc_fn": (gguf.MODEL_TENSOR.HC_FFN_FN, ".weight"), - "hc_mlp_layer.hc_pre.hc_base": (gguf.MODEL_TENSOR.HC_FFN_BASE, ".weight"), - "hc_mlp_layer.hc_pre.hc_scale": (gguf.MODEL_TENSOR.HC_FFN_SCALE, ".weight"), - "mlp.gate.weight": (gguf.MODEL_TENSOR.FFN_GATE_INP, ".weight"), - "mlp.gate.e_score_correction.bias":(gguf.MODEL_TENSOR.FFN_EXP_PROBS_B, ".bias"), - "mlp.gate_proj.weight": (gguf.MODEL_TENSOR.FFN_GATE, ".weight"), - "mlp.up_proj.weight": (gguf.MODEL_TENSOR.FFN_UP, ".weight"), - "mlp.down_proj.weight": (gguf.MODEL_TENSOR.FFN_DOWN, ".weight"), - "mlp.shared_experts.gate_proj.weight": (gguf.MODEL_TENSOR.FFN_GATE_SHEXP, ".weight"), - "mlp.shared_experts.up_proj.weight": (gguf.MODEL_TENSOR.FFN_UP_SHEXP, ".weight"), - "mlp.shared_experts.down_proj.weight": (gguf.MODEL_TENSOR.FFN_DOWN_SHEXP, ".weight"), - } - if suffix in simple: - key, sfx = simple[suffix] - return [(tn(key, bid, sfx), data_torch)] - - # kv_b_proj: split into k_b (transposed) and v_b - if suffix == "self_attn.kv_b_proj.weight": - k_b, v_b = split_kv_b_proj(data_torch, n_head, qk_nope, v_head_dim) - return [ - (tn(gguf.MODEL_TENSOR.ATTN_K_B, bid), k_b), - (tn(gguf.MODEL_TENSOR.ATTN_V_B, bid), v_b), - ] - # fused stacked experts: split gate_up into gate/up - if suffix == "mlp.experts.gate_up_proj": + if name.endswith("mlp.experts.gate_up_proj"): gate, up = split_gate_up(data_torch, moe_inter) - return [ - (tn(gguf.MODEL_TENSOR.FFN_GATE_EXP, bid), gate), - (tn(gguf.MODEL_TENSOR.FFN_UP_EXP, bid), up), - ] - if suffix == "mlp.experts.down_proj": - return [(tn(gguf.MODEL_TENSOR.FFN_DOWN_EXP, bid), data_torch)] + yield from super().modify_tensors(gate, tn(gguf.MODEL_TENSOR.FFN_GATE_EXP, bid), bid) + yield from super().modify_tensors(up, tn(gguf.MODEL_TENSOR.FFN_UP_EXP, bid), bid) + return - raise ValueError(f"Unsupported HY_V4 tensor {name!r} (suffix {suffix!r})") + # add .weight suffixes + if name.endswith("mlp.experts.down_proj") or name.endswith(".self_attn.learnable_sink_param"): + name += ".weight" + + if re.search(r"\.hc_head\.hc_head_(?:fn|base|scale)$", name): + name += ".weight" + + if re.search(r"\.hc_(?:attn|mlp)_layer\.hc_pre\.hc_(?:fn|base|scale)$", name): + name += ".weight" + + yield from super().modify_tensors(data_torch, name, bid) diff --git a/gguf-py/gguf/tensor_mapping.py b/gguf-py/gguf/tensor_mapping.py index d644d502e..d2dfeece5 100644 --- a/gguf-py/gguf/tensor_mapping.py +++ b/gguf-py/gguf/tensor_mapping.py @@ -385,6 +385,7 @@ class TensorNameMap: MODEL_TENSOR.ATTN_SINKS: ( "model.layers.{bid}.self_attn.sinks", # openai-moe "model.layers.{bid}.self_attn.attention_sink_bias", # mimov2 + "model.layers.{bid}.self_attn.learnable_sink_param", # hy-v4 ), MODEL_TENSOR.ATTN_GATE: ( @@ -392,6 +393,7 @@ class TensorNameMap: "model.layers.{bid}.linear_attn.in_proj_z", # qwen3.5 "model.layers.{bid}.self_attn.g_proj", # step3.5 head-wise attention gate "model.layers.{bid}.self_attn.output_gate", # minimax-01 + "model.layers.{bid}.self_attn.linear_gate", # hy-v4 ), # Feed-forward norm @@ -1329,6 +1331,42 @@ class TensorNameMap: "model.layers.{bid}.self_attn.index_q_norm", # MSA ), + MODEL_TENSOR.HC_ATTN_FN: ( + "model.layers.{bid}.hc_attn_layer.hc_pre.hc_fn", # hy-v4 + ), + + MODEL_TENSOR.HC_ATTN_BASE: ( + "model.layers.{bid}.hc_attn_layer.hc_pre.hc_base", # hy-v4 + ), + + MODEL_TENSOR.HC_ATTN_SCALE: ( + "model.layers.{bid}.hc_attn_layer.hc_pre.hc_scale", # hy-v4 + ), + + MODEL_TENSOR.HC_FFN_FN: ( + "model.layers.{bid}.hc_mlp_layer.hc_pre.hc_fn", # hy-v4 + ), + + MODEL_TENSOR.HC_FFN_BASE: ( + "model.layers.{bid}.hc_mlp_layer.hc_pre.hc_base", # hy-v4 + ), + + MODEL_TENSOR.HC_FFN_SCALE: ( + "model.layers.{bid}.hc_mlp_layer.hc_pre.hc_scale", # hy-v4 + ), + + MODEL_TENSOR.HC_HEAD_FN: ( + "model.hc_head.hc_head_fn", # hy-v4 + ), + + MODEL_TENSOR.HC_HEAD_BASE: ( + "model.hc_head.hc_head_base", # hy-v4 + ), + + MODEL_TENSOR.HC_HEAD_SCALE: ( + "model.hc_head.hc_head_scale", # hy-v4 + ), + ############################################################################ # TODO: these do not belong to block_mappings_cfg - move them to mappings_cfg MODEL_TENSOR.ENC_OUTPUT_NORM: (