Support GlmMoeDsaForCausalLM (#18521)

Signed-off-by: Xinyuan Tong <xinyuantong.cs@gmail.com>
Signed-off-by: BBuf <1182563586@qq.com>
Co-authored-by: Xiaoyu Zhang <35585791+BBuf@users.noreply.github.com>
Co-authored-by: BBuf <1182563586@qq.com>
This commit is contained in:
Xinyuan Tong
2026-02-10 15:20:10 +08:00
committed by GitHub
co-authored by Xiaoyu Zhang BBuf
parent e8a2c13380
commit 398b81f78c
3 changed files with 22 additions and 7 deletions
+6 -5
View File
@@ -61,6 +61,7 @@ def is_deepseek_nsa(config: PretrainedConfig) -> bool:
"DeepseekV3ForCausalLMNextN",
"MistralLarge3ForCausalLM",
"PixtralForConditionalGeneration",
"GlmMoeDsaForCausalLM",
]
and getattr(config, "index_topk", None) is not None
)
@@ -271,10 +272,10 @@ class ModelConfig:
def _config_draft_model(self):
is_draft_model = self.is_draft_model
if (
is_draft_model
and self.hf_config.architectures[0] == "DeepseekV3ForCausalLM"
):
if is_draft_model and self.hf_config.architectures[0] in [
"DeepseekV3ForCausalLM",
"GlmMoeDsaForCausalLM",
]:
self.hf_config.architectures[0] = "DeepseekV3ForCausalLMNextN"
if is_draft_model and self.hf_config.architectures[0] in [
@@ -411,7 +412,6 @@ class ModelConfig:
"swa_v_head_dim",
self.v_head_dim,
)
# FIXME: temporary special judge for MLA architecture
if (
"DeepseekV2ForCausalLM" in self.hf_config.architectures
@@ -419,6 +419,7 @@ class ModelConfig:
or "DeepseekV3ForCausalLM" in self.hf_config.architectures
or "DeepseekV3ForCausalLMNextN" in self.hf_config.architectures
or "Glm4MoeLiteForCausalLM" in self.hf_config.architectures
or "GlmMoeDsaForCausalLM" in self.hf_config.architectures
or "LongcatFlashForCausalLM" in self.hf_config.architectures
or "LongcatFlashForCausalLMNextN" in self.hf_config.architectures
or "DotsVLMForCausalLM" in self.hf_config.architectures