[Fix] GLM 4.7 + NVFP4 + MTP (#17166)
This commit is contained in:
@@ -594,6 +594,44 @@ def filter_duplicate_safetensors_files(
|
||||
return hf_weights_files
|
||||
|
||||
|
||||
def maybe_add_mtp_safetensors(
|
||||
hf_weights_files: List[str], hf_folder: str, index_file: str, hf_config
|
||||
) -> List[str]:
|
||||
"""
|
||||
Auto-detect and add mtp.safetensors for GLM4Moe MTP/NextN models if:
|
||||
1. mtp.safetensors exists in the model directory
|
||||
2. mtp.safetensors is NOT in the index (checkpoint packaging bug)
|
||||
3. Model architecture is Glm4MoeForCausalLM with num_nextn_predict_layers > 0
|
||||
|
||||
This works around incorrectly packaged FP4 checkpoints like
|
||||
baseten-admin/glm-4.7-fp4 where mtp.safetensors exists but
|
||||
isn't referenced in model.safetensors.index.json.
|
||||
"""
|
||||
# Only apply for GLM4Moe architecture with nextn layers
|
||||
arch = getattr(hf_config, "architectures", [None])[0]
|
||||
num_nextn_layers = getattr(hf_config, "num_nextn_predict_layers", 0)
|
||||
if not (
|
||||
arch in ["Glm4MoeForCausalLM", "Glm4MoeForCausalLMNextN"]
|
||||
and num_nextn_layers > 0
|
||||
):
|
||||
return hf_weights_files
|
||||
|
||||
# Check if mtp.safetensors exists and is not already in the file list
|
||||
mtp_path = os.path.join(hf_folder, "mtp.safetensors")
|
||||
if not os.path.isfile(mtp_path) or mtp_path in hf_weights_files:
|
||||
return hf_weights_files
|
||||
|
||||
# mtp.safetensors exists but not in index - this is a bug
|
||||
logger.warning(
|
||||
f"Found mtp.safetensors but it's not referenced in {index_file}. "
|
||||
f"This is a checkpoint packaging bug. Auto-adding it for loading. "
|
||||
f"Please report this to the checkpoint provider."
|
||||
)
|
||||
|
||||
# Add it to the files list
|
||||
return hf_weights_files + [mtp_path]
|
||||
|
||||
|
||||
def filter_files_not_needed_for_inference(hf_weights_files: List[str]) -> List[str]:
|
||||
"""
|
||||
Exclude files that are not needed for inference.
|
||||
|
||||
Reference in New Issue
Block a user